Publish party-position estimates and validation materials

This commit is contained in:
Armin Seimel
2026-08-13 15:26:21 +00:00
commit 7666224565
119 changed files with 87813 additions and 0 deletions
+10
View File
@@ -0,0 +1,10 @@
# party2d_estimates agent notes
- For repositories on `git.seimel.app`, use the existing Tea/Gitea token from `~/.config/tea/config.yml` for authenticated `git push`/`git fetch` operations.
- Do not print, log, or expose the token.
- Prefer token-authenticated HTTPS over unauthenticated HTTPS or SSH when pushing to `git.seimel.app`.
- Python is forbidden for this project. Public workflow code and documentation must use R, Julia, Stan, and shell only; do not add Python scripts or Python references.
- `data-setup/run_data_setup.sh` is intentionally a no-option public entry point. It downloads script-accessible sources, checks local raw files, rebuilds generated inputs under `_local/`, compares them to committed `data/`, and never replaces committed inputs automatically.
- This is a public release repository. Commit only publication-ready code, data, documentation, figures, and release metadata. Keep working notes, private correspondence, assessment records, decision logs, draft response material, and non-public planning outside this repository.
- Before every public push, run `bash scripts/check_public_content.sh`. Install the version-controlled pre-push guard with `bash scripts/install_public_guard.sh` in each local clone.
- If material was committed here in error, stop publication work. Removing it in a later commit is insufficient: create a clean replacement history or repository before resuming publication.
+19
View File
@@ -0,0 +1,19 @@
CC BY 4.0
Creative Commons Attribution 4.0 International
This repository is released under the Creative Commons Attribution 4.0
International License (CC BY 4.0), except where third-party source terms
require otherwise.
You are free to share and adapt the licensed material for any purpose,
including commercial use, provided that appropriate credit is given, a link to
the license is provided, and any changes are indicated.
This license applies to the dataset, metadata, documentation, diagnostics, and
repository workflow materials created for this release. It does not relicense
third-party source data used to construct the release inputs; those sources
remain governed by their own terms.
License deed: https://creativecommons.org/licenses/by/4.0/
Legal code: https://creativecommons.org/licenses/by/4.0/legalcode
+663
View File
@@ -0,0 +1,663 @@
# This file is machine-generated - editing it directly is not advised
julia_version = "1.11.3"
manifest_format = "2.0"
project_hash = "5d7dd35b02e3ab85cb2ccc99d83afe86c825420e"
[[deps.AbstractFFTs]]
deps = ["LinearAlgebra"]
git-tree-sha1 = "d92ad398961a3ed262d8bf04a1a2b8340f915fef"
uuid = "621f4979-c628-5d54-868e-fcf4e3e8185c"
version = "1.5.0"
[deps.AbstractFFTs.extensions]
AbstractFFTsChainRulesCoreExt = "ChainRulesCore"
AbstractFFTsTestExt = "Test"
[deps.AbstractFFTs.weakdeps]
ChainRulesCore = "d360d2e6-b24c-11e9-a2a3-2a2ae2dbcce4"
Test = "8dfed614-e22c-5e08-85e1-65c5234f0b40"
[[deps.AliasTables]]
deps = ["PtrArrays", "Random"]
git-tree-sha1 = "9876e1e164b144ca45e9e3198d0b689cadfed9ff"
uuid = "66dad0bd-aa9a-41b7-9441-69ab47430ed8"
version = "1.1.3"
[[deps.ArgTools]]
uuid = "0dad84c5-d112-42e6-8d28-ef12dabb789f"
version = "1.1.2"
[[deps.Artifacts]]
uuid = "56f22d72-fd6d-98f1-02f0-08ddc0907c33"
version = "1.11.0"
[[deps.Base64]]
uuid = "2a0f44e3-6c83-55bd-87e4-b1978d98bd5f"
version = "1.11.0"
[[deps.CSV]]
deps = ["CodecZlib", "Dates", "FilePathsBase", "InlineStrings", "Mmap", "Parsers", "PooledArrays", "PrecompileTools", "SentinelArrays", "Tables", "Unicode", "WeakRefStrings", "WorkerUtilities"]
git-tree-sha1 = "8d8e0b0f350b8e1c91420b5e64e5de774c2f0f4d"
uuid = "336ed68f-0bac-5ca0-87d4-7b16caf5d00b"
version = "0.10.16"
[[deps.CategoricalArrays]]
deps = ["Compat", "DataAPI", "Future", "Missings", "Printf", "Requires", "Statistics", "Unicode"]
git-tree-sha1 = "20ff1463035a170b25eba2ef9823bb9ad51635e4"
uuid = "324d7699-5711-5eae-9e2f-1d82baa6b597"
version = "1.1.1"
[deps.CategoricalArrays.extensions]
CategoricalArraysArrowExt = "Arrow"
CategoricalArraysJSONExt = "JSON"
CategoricalArraysRecipesBaseExt = "RecipesBase"
CategoricalArraysSentinelArraysExt = "SentinelArrays"
CategoricalArraysStatsBaseExt = "StatsBase"
CategoricalArraysStructTypesExt = "StructTypes"
[deps.CategoricalArrays.weakdeps]
Arrow = "69666777-d1a9-59fb-9406-91d4454c9d45"
JSON = "682c06a0-de6a-54ab-a142-c8b1cf79cde6"
RecipesBase = "3cdcf5f2-1ef4-517c-9805-6587b60abb01"
SentinelArrays = "91c51154-3ec4-41a3-a24f-3f23e20d615c"
StatsBase = "2913bbd2-ae8a-5f71-8c99-4fb6c76f3a91"
StructTypes = "856f2bd8-1eba-4b0a-8007-ebc267875bd4"
[[deps.CodecZlib]]
deps = ["TranscodingStreams", "Zlib_jll"]
git-tree-sha1 = "962834c22b66e32aa10f7611c08c8ca4e20749a9"
uuid = "944b1d66-785c-5afd-91f1-9de20f533193"
version = "0.7.8"
[[deps.Compat]]
deps = ["TOML", "UUIDs"]
git-tree-sha1 = "9d8a54ce4b17aa5bdce0ea5c34bc5e7c340d16ad"
uuid = "34da2185-b29b-5c13-b0c7-acf172513d20"
version = "4.18.1"
weakdeps = ["Dates", "LinearAlgebra"]
[deps.Compat.extensions]
CompatLinearAlgebraExt = "LinearAlgebra"
[[deps.CompatHelperLocal]]
deps = ["Pkg"]
git-tree-sha1 = "f239f702063ca2fc50844c6cf27b4d497652b474"
uuid = "5224ae11-6099-4aaa-941d-3aab004bd678"
version = "0.1.29"
[[deps.CompilerSupportLibraries_jll]]
deps = ["Artifacts", "Libdl"]
uuid = "e66e0078-7015-5450-92f7-15fbd957f2ae"
version = "1.1.1+0"
[[deps.Crayons]]
git-tree-sha1 = "249fe38abf76d48563e2f4556bebd215aa317e15"
uuid = "a8cc5b0e-0ffa-5ad4-8c14-923d3ee1735f"
version = "4.1.1"
[[deps.DataAPI]]
git-tree-sha1 = "abe83f3a2f1b857aac70ef8b269080af17764bbe"
uuid = "9a962f9c-6df0-11e9-0e5d-c546b8b5ee8a"
version = "1.16.0"
[[deps.DataFrames]]
deps = ["Compat", "DataAPI", "DataStructures", "Future", "InlineStrings", "InvertedIndices", "IteratorInterfaceExtensions", "LinearAlgebra", "Markdown", "Missings", "PooledArrays", "PrecompileTools", "PrettyTables", "Printf", "Random", "Reexport", "SentinelArrays", "SortingAlgorithms", "Statistics", "TableTraits", "Tables", "Unicode"]
git-tree-sha1 = "5fab31e2e01e70ad66e3e24c968c264d1cf166d6"
uuid = "a93c6f00-e57d-5684-b7b6-d8193f3e46c0"
version = "1.8.2"
[[deps.DataStructures]]
deps = ["OrderedCollections"]
git-tree-sha1 = "6fb53a69613a0b2b68a0d12671717d307ab8b24e"
uuid = "864edb3b-99cc-5e75-8d2d-829cb0a9cfe8"
version = "0.19.5"
[[deps.DataValueInterfaces]]
git-tree-sha1 = "bfc1187b79289637fa0ef6d4436ebdfe6905cbd6"
uuid = "e2d170a0-9d28-54be-80f0-106bbe20a464"
version = "1.0.0"
[[deps.Dates]]
deps = ["Printf"]
uuid = "ade2ca70-3891-5945-98fb-dc099432e06a"
version = "1.11.0"
[[deps.DelimitedFiles]]
deps = ["Mmap"]
git-tree-sha1 = "9e2f36d3c96a820c678f2f1f1782582fcf685bae"
uuid = "8bb1440f-4735-579b-a4ab-409b98df4dab"
version = "1.9.1"
[[deps.Distributed]]
deps = ["Random", "Serialization", "Sockets"]
uuid = "8ba89e20-285c-5b6f-9357-94700520ee1b"
version = "1.11.0"
[[deps.DocStringExtensions]]
git-tree-sha1 = "7442a5dfe1ebb773c29cc2962a8980f47221d76c"
uuid = "ffbed154-4ef7-542d-bbb7-c09d3a79fcae"
version = "0.9.5"
[[deps.Downloads]]
deps = ["ArgTools", "FileWatching", "LibCURL", "NetworkOptions"]
uuid = "f43a241f-c20a-4ad4-852c-f6b1247861c6"
version = "1.6.0"
[[deps.FFTW]]
deps = ["AbstractFFTs", "FFTW_jll", "Libdl", "LinearAlgebra", "MKL_jll", "Preferences", "Reexport"]
git-tree-sha1 = "97f08406df914023af55ade2f843c39e99c5d969"
uuid = "7a1cc6ca-52ef-59f5-83cd-3a7055c09341"
version = "1.10.0"
[[deps.FFTW_jll]]
deps = ["Artifacts", "JLLWrappers", "Libdl"]
git-tree-sha1 = "6866aec60ef98e3164cd8d6855225684207e9dff"
uuid = "f5851436-0d7a-5f13-b9de-f02708fd171a"
version = "3.3.12+0"
[[deps.FilePathsBase]]
deps = ["Compat", "Dates"]
git-tree-sha1 = "3bab2c5aa25e7840a4b065805c0cdfc01f3068d2"
uuid = "48062228-2e41-5def-b9a4-89aafe57970f"
version = "0.9.24"
weakdeps = ["Mmap", "Test"]
[deps.FilePathsBase.extensions]
FilePathsBaseMmapExt = "Mmap"
FilePathsBaseTestExt = "Test"
[[deps.FileWatching]]
uuid = "7b1f6079-737a-58dc-b8bc-7a2ca5c1b5ee"
version = "1.11.0"
[[deps.Future]]
deps = ["Random"]
uuid = "9fa8497b-333b-5362-9e8d-4d0656e87820"
version = "1.11.0"
[[deps.HypergeometricFunctions]]
deps = ["LinearAlgebra", "OpenLibm_jll", "SpecialFunctions"]
git-tree-sha1 = "68c173f4f449de5b438ee67ed0c9c748dc31a2ec"
uuid = "34004b35-14d8-5ef3-9330-4cdb6864b03a"
version = "0.3.28"
[[deps.InlineStrings]]
git-tree-sha1 = "8f3d257792a522b4601c24a577954b0a8cd7334d"
uuid = "842dd82b-1e85-43dc-bf29-5d0ee9dffc48"
version = "1.4.5"
[deps.InlineStrings.extensions]
ArrowTypesExt = "ArrowTypes"
ParsersExt = "Parsers"
[deps.InlineStrings.weakdeps]
ArrowTypes = "31f734f8-188a-4ce0-8406-c8a06bd891cd"
Parsers = "69de0a69-1ddd-5017-9359-2bf0b02dc9f0"
[[deps.IntelOpenMP_jll]]
deps = ["Artifacts", "JLLWrappers", "LazyArtifacts", "Libdl"]
git-tree-sha1 = "ec1debd61c300961f98064cfb21287613ad7f303"
uuid = "1d5cc7b8-4909-519e-a0f8-d0f5ad9712d0"
version = "2025.2.0+0"
[[deps.InteractiveUtils]]
deps = ["Markdown"]
uuid = "b77e0a4c-d291-57a0-90e8-8db25a27a240"
version = "1.11.0"
[[deps.InvertedIndices]]
git-tree-sha1 = "6da3c4316095de0f5ee2ebd875df8721e7e0bdbe"
uuid = "41ab1584-1d38-5bbf-9106-f11c6c58b48f"
version = "1.3.1"
[[deps.IrrationalConstants]]
git-tree-sha1 = "b2d91fe939cae05960e760110b328288867b5758"
uuid = "92d709cd-6900-40b7-9082-c6be49f344b6"
version = "0.2.6"
[[deps.IteratorInterfaceExtensions]]
git-tree-sha1 = "a3f24677c21f5bbe9d2a714f95dcd58337fb2856"
uuid = "82899510-4779-5014-852e-03e436cf321d"
version = "1.0.0"
[[deps.JLLWrappers]]
deps = ["Artifacts", "Preferences"]
git-tree-sha1 = "7204148362dafe5fe6a273f855b8ccbe4df8173e"
uuid = "692b3bcd-3c85-4b1f-b108-f13ce0eb3210"
version = "1.8.0"
[[deps.JSON]]
deps = ["Dates", "Mmap", "Parsers", "Unicode"]
git-tree-sha1 = "31e996f0a15c7b280ba9f76636b3ff9e2ae58c9a"
uuid = "682c06a0-de6a-54ab-a142-c8b1cf79cde6"
version = "0.21.4"
[[deps.LaTeXStrings]]
git-tree-sha1 = "dda21b8cbd6a6c40d9d02a73230f9d70fed6918c"
uuid = "b964fa9f-0449-5b57-a5c2-d3ea65f4040f"
version = "1.4.0"
[[deps.LazyArtifacts]]
deps = ["Artifacts", "Pkg"]
uuid = "4af54fe1-eca0-43a8-85a7-787d91b784e3"
version = "1.11.0"
[[deps.LibCURL]]
deps = ["LibCURL_jll", "MozillaCACerts_jll"]
uuid = "b27032c2-a3e7-50c8-80cd-2d36dbcbfd21"
version = "0.6.4"
[[deps.LibCURL_jll]]
deps = ["Artifacts", "LibSSH2_jll", "Libdl", "MbedTLS_jll", "Zlib_jll", "nghttp2_jll"]
uuid = "deac9b47-8bc7-5906-a0fe-35ac56dc84c0"
version = "8.6.0+0"
[[deps.LibGit2]]
deps = ["Base64", "LibGit2_jll", "NetworkOptions", "Printf", "SHA"]
uuid = "76f85450-5226-5b5a-8eaa-529ad045b433"
version = "1.11.0"
[[deps.LibGit2_jll]]
deps = ["Artifacts", "LibSSH2_jll", "Libdl", "MbedTLS_jll"]
uuid = "e37daf67-58a4-590a-8e99-b0245dd2ffc5"
version = "1.7.2+0"
[[deps.LibSSH2_jll]]
deps = ["Artifacts", "Libdl", "MbedTLS_jll"]
uuid = "29816b5a-b9ab-546f-933c-edad1886dfa8"
version = "1.11.0+1"
[[deps.Libdl]]
uuid = "8f399da3-3557-5675-b5ff-fb832c97cbdb"
version = "1.11.0"
[[deps.LinearAlgebra]]
deps = ["Libdl", "OpenBLAS_jll", "libblastrampoline_jll"]
uuid = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e"
version = "1.11.0"
[[deps.LogExpFunctions]]
deps = ["DocStringExtensions", "IrrationalConstants", "LinearAlgebra"]
git-tree-sha1 = "bba2d9aa057d8f126415de240573e86a8f39d2a1"
uuid = "2ab3a3ac-af41-5b50-aa03-7779005ae688"
version = "1.0.1"
[deps.LogExpFunctions.extensions]
LogExpFunctionsChainRulesCoreExt = "ChainRulesCore"
LogExpFunctionsChangesOfVariablesExt = "ChangesOfVariables"
LogExpFunctionsInverseFunctionsExt = "InverseFunctions"
[deps.LogExpFunctions.weakdeps]
ChainRulesCore = "d360d2e6-b24c-11e9-a2a3-2a2ae2dbcce4"
ChangesOfVariables = "9e997f8a-9a97-42d5-a9f1-ce6bfc15e2c0"
InverseFunctions = "3587e190-3f89-42d0-90ee-14403ec27112"
[[deps.Logging]]
uuid = "56ddb016-857b-54e1-b83d-db4d58db5568"
version = "1.11.0"
[[deps.MKL_jll]]
deps = ["Artifacts", "IntelOpenMP_jll", "JLLWrappers", "LazyArtifacts", "Libdl", "oneTBB_jll"]
git-tree-sha1 = "282cadc186e7b2ae0eeadbd7a4dffed4196ae2aa"
uuid = "856f044c-d86e-5d09-b602-aeab76dc8ba7"
version = "2025.2.0+0"
[[deps.Markdown]]
deps = ["Base64"]
uuid = "d6f4376e-aef5-505a-96c1-9c027394607a"
version = "1.11.0"
[[deps.MbedTLS_jll]]
deps = ["Artifacts", "Libdl"]
uuid = "c8ffd9c3-330d-5841-b78e-0817d7145fa1"
version = "2.28.6+0"
[[deps.Missings]]
deps = ["DataAPI"]
git-tree-sha1 = "ec4f7fbeab05d7747bdf98eb74d130a2a2ed298d"
uuid = "e1d29d7a-bbdc-5cf2-9ac0-f12de2c33e28"
version = "1.2.0"
[[deps.Mmap]]
uuid = "a63ad114-7e13-5084-954f-fe012c677804"
version = "1.11.0"
[[deps.MozillaCACerts_jll]]
uuid = "14a3606d-f60d-562e-9121-12d972cd8159"
version = "2023.12.12"
[[deps.NamedTupleTools]]
git-tree-sha1 = "90914795fc59df44120fe3fff6742bb0d7adb1d0"
uuid = "d9ec5142-1e00-5aa0-9d6a-321866360f50"
version = "0.14.3"
[[deps.NetworkOptions]]
uuid = "ca575930-c2e3-43a9-ace4-1e988b2c1908"
version = "1.2.0"
[[deps.OpenBLAS_jll]]
deps = ["Artifacts", "CompilerSupportLibraries_jll", "Libdl"]
uuid = "4536629a-c528-5b80-bd46-f80d51c5b363"
version = "0.3.27+1"
[[deps.OpenLibm_jll]]
deps = ["Artifacts", "Libdl"]
uuid = "05823500-19ac-5b8b-9628-191a04bc5112"
version = "0.8.1+2"
[[deps.OpenSpecFun_jll]]
deps = ["Artifacts", "CompilerSupportLibraries_jll", "JLLWrappers", "Libdl"]
git-tree-sha1 = "1346c9208249809840c91b26703912dff463d335"
uuid = "efe28fd5-8261-553b-a9e1-b2916fc3738e"
version = "0.5.6+0"
[[deps.OrderedCollections]]
git-tree-sha1 = "94ba93778373a53bfd5a0caaf7d809c445292ff4"
uuid = "bac558e1-5e72-5ebc-8fee-abe8a469f55d"
version = "1.8.2"
[[deps.Parameters]]
deps = ["OrderedCollections", "UnPack"]
git-tree-sha1 = "34c0e9ad262e5f7fc75b10a9952ca7692cfc5fbe"
uuid = "d96e819e-fc66-5662-9728-84c9c7592b0a"
version = "0.12.3"
[[deps.Parsers]]
deps = ["Dates", "PrecompileTools", "UUIDs"]
git-tree-sha1 = "468dbe2b510c876dc091b2c74ed52c7c34f48b9b"
uuid = "69de0a69-1ddd-5017-9359-2bf0b02dc9f0"
version = "2.8.5"
[[deps.Pkg]]
deps = ["Artifacts", "Dates", "Downloads", "FileWatching", "LibGit2", "Libdl", "Logging", "Markdown", "Printf", "Random", "SHA", "TOML", "Tar", "UUIDs", "p7zip_jll"]
uuid = "44cfe95a-1eb2-52ea-b672-e2afdf69b78f"
version = "1.11.0"
weakdeps = ["REPL"]
[deps.Pkg.extensions]
REPLExt = "REPL"
[[deps.PooledArrays]]
deps = ["DataAPI", "Future"]
git-tree-sha1 = "36d8b4b899628fb92c2749eb488d884a926614d3"
uuid = "2dfb63ee-cc39-5dd5-95bd-886bf059d720"
version = "1.4.3"
[[deps.PrecompileTools]]
deps = ["Preferences"]
git-tree-sha1 = "5aa36f7049a63a1528fe8f7c3f2113413ffd4e1f"
uuid = "aea7be01-6a6a-4083-8856-8a6e6704d82a"
version = "1.2.1"
[[deps.Preferences]]
deps = ["TOML"]
git-tree-sha1 = "8b770b60760d4451834fe79dd483e318eee709c4"
uuid = "21216c6a-2e73-6563-6e65-726566657250"
version = "1.5.2"
[[deps.PrettyTables]]
deps = ["Crayons", "LaTeXStrings", "Markdown", "PrecompileTools", "Printf", "REPL", "Reexport", "StringManipulation", "Tables"]
git-tree-sha1 = "624de6279ab7d94fc9f672f0068107eb6619732c"
uuid = "08abe8d2-0d0c-5749-adfa-8a2ac140af0d"
version = "3.3.2"
[deps.PrettyTables.extensions]
PrettyTablesTypstryExt = "Typstry"
[deps.PrettyTables.weakdeps]
Typstry = "f0ed7684-a786-439e-b1e3-3b82803b501e"
[[deps.Printf]]
deps = ["Unicode"]
uuid = "de0858da-6303-5e67-8744-51eddeeeb8d7"
version = "1.11.0"
[[deps.ProgressMeter]]
deps = ["Distributed", "Printf"]
git-tree-sha1 = "fbb92c6c56b34e1a2c4c36058f68f332bec840e7"
uuid = "92933f4c-e287-5a05-a399-4b506db050ca"
version = "1.11.0"
[[deps.PtrArrays]]
git-tree-sha1 = "4fbbafbc6251b883f4d2705356f3641f3652a7fe"
uuid = "43287f4e-b6f4-7ad1-bb20-aadabca52c3d"
version = "1.4.0"
[[deps.REPL]]
deps = ["InteractiveUtils", "Markdown", "Sockets", "StyledStrings", "Unicode"]
uuid = "3fa0cd96-eef1-5676-8a61-b3b8758bbffb"
version = "1.11.0"
[[deps.Random]]
deps = ["SHA"]
uuid = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c"
version = "1.11.0"
[[deps.Reexport]]
git-tree-sha1 = "45e428421666073eab6f2da5c9d310d99bb12f9b"
uuid = "189a3867-3050-52da-a836-e630ba90ab69"
version = "1.2.2"
[[deps.Requires]]
deps = ["UUIDs"]
git-tree-sha1 = "62389eeff14780bfe55195b7204c0d8738436d64"
uuid = "ae029012-a4dd-5104-9daa-d747884805df"
version = "1.3.1"
[[deps.Rmath]]
deps = ["Random", "Rmath_jll"]
git-tree-sha1 = "5b3d50eb374cea306873b371d3f8d3915a018f0b"
uuid = "79098fc4-a85e-5d69-aa6a-4863f24498fa"
version = "0.9.0"
[[deps.Rmath_jll]]
deps = ["Artifacts", "JLLWrappers", "Libdl"]
git-tree-sha1 = "58cdd8fb2201a6267e1db87ff148dd6c1dbd8ad8"
uuid = "f50d1b31-88e8-58de-be2c-1cc44531875f"
version = "0.5.1+0"
[[deps.SHA]]
uuid = "ea8e919c-243c-51af-8825-aaa63cd721ce"
version = "0.7.0"
[[deps.SentinelArrays]]
deps = ["Dates", "Random"]
git-tree-sha1 = "084c47c7c5ce5cfecefa0a98dff69eb3646b5a80"
uuid = "91c51154-3ec4-41a3-a24f-3f23e20d615c"
version = "1.4.10"
[[deps.Serialization]]
uuid = "9e88b42a-f829-5b0c-bbe9-9e923198166b"
version = "1.11.0"
[[deps.Sockets]]
uuid = "6462fe0b-24de-5631-8697-dd941f90decc"
version = "1.11.0"
[[deps.SortingAlgorithms]]
deps = ["DataStructures"]
git-tree-sha1 = "64d974c2e6fdf07f8155b5b2ca2ffa9069b608d9"
uuid = "a2af1166-a08f-5f64-846c-94a0d3cef48c"
version = "1.2.2"
[[deps.SparseArrays]]
deps = ["Libdl", "LinearAlgebra", "Random", "Serialization", "SuiteSparse_jll"]
uuid = "2f01184e-e22b-5df5-ae63-d93ebab69eaf"
version = "1.11.0"
[[deps.SpecialFunctions]]
deps = ["IrrationalConstants", "LogExpFunctions", "OpenLibm_jll", "OpenSpecFun_jll"]
git-tree-sha1 = "6547cbdd8ce32efba0d21c5a40fa96d1a3548f9f"
uuid = "276daf66-3868-5448-9aa4-cd146d93841b"
version = "2.8.0"
[deps.SpecialFunctions.extensions]
SpecialFunctionsChainRulesCoreExt = "ChainRulesCore"
[deps.SpecialFunctions.weakdeps]
ChainRulesCore = "d360d2e6-b24c-11e9-a2a3-2a2ae2dbcce4"
[[deps.StanBase]]
deps = ["CSV", "DataFrames", "DelimitedFiles", "Distributed", "DocStringExtensions", "JSON", "NamedTupleTools", "OrderedCollections", "Parameters", "Random", "Unicode"]
git-tree-sha1 = "fac478e96c6a32f08af58f709b07f5ee722b3140"
uuid = "d0ee94f6-a23d-54aa-bbe9-7f572d6da7f5"
version = "4.12.4"
[[deps.StanSample]]
deps = ["CSV", "CompatHelperLocal", "DataFrames", "DelimitedFiles", "DocStringExtensions", "JSON", "LazyArtifacts", "NamedTupleTools", "OrderedCollections", "Parameters", "Random", "Reexport", "Requires", "Serialization", "StanBase", "TableOperations", "Tables", "Unicode"]
git-tree-sha1 = "0e6f7c729879a416fd7b27cfd81b2b1f6e08586e"
uuid = "c1514b29-d3a0-5178-b312-660c88baa699"
version = "7.10.3"
[deps.StanSample.extensions]
AxisKeysExt = "AxisKeys"
InferenceObjectsExt = "InferenceObjects"
MCMCChainsExt = "MCMCChains"
MonteCarloMeasurementsExt = "MonteCarloMeasurements"
[deps.StanSample.weakdeps]
AxisKeys = "94b1ba4f-4ee9-5380-92f1-94cde586c3c5"
InferenceObjects = "b5cf5a8d-e756-4ee3-b014-01d49d192c00"
MCMCChains = "c7f686f2-ff18-58e9-bc7b-31028e88f75d"
MonteCarloMeasurements = "0987c9cc-fe09-11e8-30f0-b96dd679fdca"
[[deps.Statistics]]
deps = ["LinearAlgebra"]
git-tree-sha1 = "ae3bb1eb3bba077cd276bc5cfc337cc65c3075c0"
uuid = "10745b16-79ce-11e8-11f9-7d13ad32a3b2"
version = "1.11.1"
weakdeps = ["SparseArrays"]
[deps.Statistics.extensions]
SparseArraysExt = ["SparseArrays"]
[[deps.StatsAPI]]
deps = ["LinearAlgebra"]
git-tree-sha1 = "178ed29fd5b2a2cfc3bd31c13375ae925623ff36"
uuid = "82ae8749-77ed-4fe6-ae5f-f523153014b0"
version = "1.8.0"
[[deps.StatsBase]]
deps = ["AliasTables", "DataAPI", "DataStructures", "IrrationalConstants", "LinearAlgebra", "LogExpFunctions", "Missings", "Printf", "Random", "SortingAlgorithms", "SparseArrays", "Statistics", "StatsAPI"]
git-tree-sha1 = "c6f18e5a52a176a383f6f6c635e0f81feed1d6d4"
uuid = "2913bbd2-ae8a-5f71-8c99-4fb6c76f3a91"
version = "0.34.11"
[[deps.StatsFuns]]
deps = ["HypergeometricFunctions", "IrrationalConstants", "LogExpFunctions", "Reexport", "Rmath", "SpecialFunctions"]
git-tree-sha1 = "3f4e1d24289cd974e089c617b1472311a2b1feab"
uuid = "4c63d2b9-4356-54db-8cca-17b64c39e42c"
version = "2.1.0"
[deps.StatsFuns.extensions]
StatsFunsChainRulesCoreExt = "ChainRulesCore"
StatsFunsInverseFunctionsExt = "InverseFunctions"
[deps.StatsFuns.weakdeps]
ChainRulesCore = "d360d2e6-b24c-11e9-a2a3-2a2ae2dbcce4"
InverseFunctions = "3587e190-3f89-42d0-90ee-14403ec27112"
[[deps.StringManipulation]]
deps = ["PrecompileTools"]
git-tree-sha1 = "d05693d339e37d6ab134c5ab53c29fce5ee5d7d5"
uuid = "892a3eda-7b42-436c-8928-eab12a02cf0e"
version = "0.4.4"
[[deps.StyledStrings]]
uuid = "f489334b-da3d-4c2e-b8f0-e476e12c162b"
version = "1.11.0"
[[deps.SuiteSparse_jll]]
deps = ["Artifacts", "Libdl", "libblastrampoline_jll"]
uuid = "bea87d4a-7f5b-5778-9afe-8cc45184846c"
version = "7.7.0+0"
[[deps.TOML]]
deps = ["Dates"]
uuid = "fa267f1f-6049-4f14-aa54-33bafae1ed76"
version = "1.0.3"
[[deps.TableOperations]]
deps = ["SentinelArrays", "Tables", "Test"]
git-tree-sha1 = "e383c87cf2a1dc41fa30c093b2a19877c83e1bc1"
uuid = "ab02a1b2-a7df-11e8-156e-fb1833f50b87"
version = "1.2.0"
[[deps.TableTraits]]
deps = ["IteratorInterfaceExtensions"]
git-tree-sha1 = "c06b2f539df1c6efa794486abfb6ed2022561a39"
uuid = "3783bdb8-4a98-5b6b-af9a-565f29a5fe9c"
version = "1.0.1"
[[deps.Tables]]
deps = ["DataAPI", "DataValueInterfaces", "IteratorInterfaceExtensions", "OrderedCollections", "TableTraits"]
git-tree-sha1 = "f2c1efbc8f3a609aadf318094f8fc5204bdaf344"
uuid = "bd369af6-aec1-5ad0-b16a-f7cc5008161c"
version = "1.12.1"
[[deps.Tar]]
deps = ["ArgTools", "SHA"]
uuid = "a4e569a6-e804-4fa4-b0f3-eef7a1d5b13e"
version = "1.10.0"
[[deps.Test]]
deps = ["InteractiveUtils", "Logging", "Random", "Serialization"]
uuid = "8dfed614-e22c-5e08-85e1-65c5234f0b40"
version = "1.11.0"
[[deps.TranscodingStreams]]
git-tree-sha1 = "0c45878dcfdcfa8480052b6ab162cdd138781742"
uuid = "3bb67fe8-82b1-5028-8e26-92a6c54297fa"
version = "0.11.3"
[[deps.UUIDs]]
deps = ["Random", "SHA"]
uuid = "cf7118a7-6976-5b1a-9a39-7adc72f591a4"
version = "1.11.0"
[[deps.UnPack]]
git-tree-sha1 = "387c1f73762231e86e0c9c5443ce3b4a0a9a0c2b"
uuid = "3a884ed6-31ef-47d7-9d2a-63182c4928ed"
version = "1.0.2"
[[deps.Unicode]]
uuid = "4ec0a83e-493e-50e2-b9ac-8f72acf5a8f5"
version = "1.11.0"
[[deps.WeakRefStrings]]
deps = ["DataAPI", "InlineStrings", "Parsers"]
git-tree-sha1 = "0716e01c3b40413de5dedbc9c5c69f27cddfddfc"
uuid = "ea10d353-3f73-51f8-a26c-33c1cb351aa5"
version = "1.4.3"
[[deps.WorkerUtilities]]
git-tree-sha1 = "cd1659ba0d57b71a464a29e64dbc67cfe83d54e7"
uuid = "76eceee3-57b5-4d4a-8e66-0e911cebbf60"
version = "1.6.1"
[[deps.Zlib_jll]]
deps = ["Libdl"]
uuid = "83775a58-1f1d-513f-b197-d71354ab007a"
version = "1.2.13+1"
[[deps.libblastrampoline_jll]]
deps = ["Artifacts", "Libdl"]
uuid = "8e850b90-86db-534c-a0d3-1478176c7d93"
version = "5.11.0+0"
[[deps.nghttp2_jll]]
deps = ["Artifacts", "Libdl"]
uuid = "8e850ede-7688-5339-a07c-302acd2aaf8d"
version = "1.59.0+0"
[[deps.oneTBB_jll]]
deps = ["Artifacts", "JLLWrappers", "LazyArtifacts", "Libdl"]
git-tree-sha1 = "da8c1f6eee04831f14edcfa5dae611d309807e57"
uuid = "1317d2d5-d96f-522e-a858-c73665f53c3e"
version = "2022.3.0+0"
[[deps.p7zip_jll]]
deps = ["Artifacts", "Libdl"]
uuid = "3f19e933-33d8-53b3-aaab-bd5110c3b7a0"
version = "17.4.0+2"
+23
View File
@@ -0,0 +1,23 @@
# Public repository policy
This repository is a public release surface. Every reachable commit, branch,
tag, file, attachment, issue, pull request, and release note must be safe for
unrestricted public access.
Allowed material is limited to publication-ready software, data, metadata,
documentation, figures, reproducibility outputs, and release records.
Do not place private correspondence, assessment records, decision logs,
working notes, draft response material, credentials, local paths, or other
non-public project context in this repository.
Before every push, run `bash scripts/check_public_content.sh`. Install the
pre-push guard in every clone with `bash scripts/install_public_guard.sh`.
If non-public material enters a public ref, stop publication work immediately.
A later deletion does not make the prior object private. Replace the affected
public history or repository with a clean snapshot, then verify that the old
object URLs no longer resolve before publishing again.
Branch names, tags, commit messages, release titles, release notes, issue
titles, and pull-request titles follow the same rule as file contents.
+10
View File
@@ -0,0 +1,10 @@
[deps]
CSV = "336ed68f-0bac-5ca0-87d4-7b16caf5d00b"
CategoricalArrays = "324d7699-5711-5eae-9e2f-1d82baa6b597"
DataFrames = "a93c6f00-e57d-5684-b7b6-d8193f3e46c0"
FFTW = "7a1cc6ca-52ef-59f5-83cd-3a7055c09341"
JSON = "682c06a0-de6a-54ab-a142-c8b1cf79cde6"
ProgressMeter = "92933f4c-e287-5a05-a399-4b506db050ca"
StanSample = "c1514b29-d3a0-5178-b312-660c88baa699"
StatsBase = "2913bbd2-ae8a-5f71-8c99-4fb6c76f3a91"
StatsFuns = "4c63d2b9-4356-54db-8cca-17b64c39e42c"
+156
View File
@@ -0,0 +1,156 @@
# party2d
Code and processed model inputs for generating two-dimensional party-position estimates from text and expert data. The model combines manifesto and media text indicators with expert survey placements in a Bayesian dynamic item-response framework.
## Repository contents
- `data-setup/` — source download, source-file checks, and rebuild workflow for raw files that cannot be redistributed here.
- `src/julia/` — Stan data preparation, model fitting, post-estimation, enrichment, and validation.
- `models/` — Stan model specification.
- `data/` — processed party-level inputs used by the Julia/Stan model.
- `metadata/` — data dictionary and source-support documentation.
- `diagnostics/` — repository diagnostics report regenerated after model estimation.
- `data/releases/` — release-ready data files, checksums, and diagnostics report.
Processed inputs needed by the model are included in `data/` so the estimation step can be reproduced from the model-ready data.
## Release assets
The public `v0` release is available from <https://git.seimel.app/armin/party2d/releases> and contains:
- `party_2d_election_year_panel_v0.zip` — primary election-year party-position panel.
- `party_2d_annual_model_output_v0.csv.xz` — secondary annual model output (XZ-compressed CSV).
- `party_2d_diagnostics_report_v0.pdf` — script-generated diagnostics report.
- `SHA256SUMS` — checksums for the release assets.
## Source data and redistribution
The public repository does **not** contain the original raw/source files. Several inputs are third-party datasets with their own terms of use, and the Morgan historical file is a local OCR/transcription source. Instead, this repository provides:
1. committed model-ready inputs in `data/`, sufficient for fitting the Julia/Stan model; and
2. `data-setup/`, an R/Shell workflow that downloads script-accessible sources, checks locally supplied raw sources, and rebuilds comparable model-ready inputs locally.
`data-setup/` downloads script-accessible sources and checks locally supplied sources for:
- PolDem
- PartyFacts crosswalk
- CHES family files
- POPPA from Harvard Dataverse
- Global Party Survey 2019 from Harvard Dataverse
- V-Party through the provider's download form
Two inputs require user-provided access/material:
- **Manifesto Project**: users must obtain the source data through their own Manifesto Project access.
- **Morgan historical expert data**: `morgan_positions_raw.csv` is not publicly downloadable; it can be provided on request and should be placed locally under `_local/raw/morgan/`.
The setup workflow never overwrites committed files in `data/`. See `data-setup/README.md` for exact commands and source details.
Run the full source-data setup workflow with:
```bash
bash data-setup/run_data_setup.sh
```
This downloads script-accessible source files, checks required local files, rebuilds model-ready inputs locally, and writes a comparison report. Manifesto Project requires your own provider access, and the Morgan OCR/transcription file can be provided on request.
## Running the pipeline
Run the full workflow with:
```bash
bash run_estimation.sh full
```
This checks that model-ready inputs are present, then executes the Julia/Stan workflow scripts:
```bash
bash scripts/01_prepare_data.sh # checks model-ready inputs; does not rebuild raw data
bash scripts/02_fit_model.sh
bash scripts/03_extract_estimates.sh
bash scripts/04_enrich_estimates.sh
bash scripts/05_validate_estimates.sh
```
The numbered scripts can also be run manually in that order.
The Bayesian model is computationally expensive. The production run used 4 cores on an AMD Ryzen 9 7945HX and took 60,372 seconds, approximately 16 hours 46 minutes.
If model output is already available, rebuild estimates without refitting Stan:
```bash
bash run_estimation.sh reuse
```
`reuse` verifies the model-ready inputs, then reruns post-estimation, enrichment, and validation while skipping the Stan fitting step.
To check the local setup without fitting the model, run:
```bash
bash run_estimation.sh dry-run
```
After estimation and validation have been run, regenerate the diagnostics report with:
```bash
Rscript diagnostics/generate_diagnostics.R
```
The report is written to `diagnostics/generated/diagnostics_report.pdf` and copied to `data/releases/party_2d_diagnostics_report_v0.pdf`.
## Data inputs
The model-ready inputs are included under `data/`:
- `text_data.csv`
- `expert.csv`
- `lr_data.csv`
- `union_mapping.csv`
- `party_families.csv`
Original raw source files are not redistributed. Rebuilding inputs from raw files is separate from the normal estimation workflow and never replaces committed `data/` inputs automatically.
## Output dimensions
The two position dimensions are scaled from 0 to 1:
- Economic left-right: economic left to economic right.
- Cultural cosmopolitan--traditionalist: cosmopolitan to traditionalist.
Column definitions are in `metadata/data_dictionary.csv`.
## Release files
The public release assets are:
- `party_2d_election_year_panel_v0.zip`
- `party_2d_annual_model_output_v0.csv.xz`
- `party_2d_diagnostics_report_v0.pdf`
- `SHA256SUMS`
`SHA256SUMS` hashes the release assets listed above.
## Public-content safeguard
This repository is intended only for publication-ready material. Before pushing,
run:
```bash
bash scripts/check_public_content.sh
```
To install the same audit as a local pre-push hook, run:
```bash
bash scripts/install_public_guard.sh
```
The audit checks the working public history and tracked files for non-public
material. If it fails, do not push until the issue is resolved.
## License
This release uses the Creative Commons Attribution 4.0 International License
(CC BY 4.0). The license applies to release materials created for this
repository; third-party source data remain governed by their own terms. See
`LICENSE`.
+16
View File
@@ -0,0 +1,16 @@
#!/usr/bin/env Rscript
repo_root <- normalizePath(getwd(), mustWork = TRUE)
lib <- Sys.getenv("R_LIBS_USER", file.path(repo_root, "_local", "R", "library"))
dir.create(lib, recursive = TRUE, showWarnings = FALSE)
.libPaths(c(lib, .libPaths()))
required <- c("tidyverse", "countrycode", "haven", "foreign", "jsonlite")
missing <- required[!vapply(required, requireNamespace, quietly = TRUE, FUN.VALUE = logical(1))]
if (length(missing) > 0) {
message("Installing missing R packages into ", lib, ": ", paste(missing, collapse = ", "))
install.packages(missing, repos = "https://cloud.r-project.org", lib = lib)
} else {
message("R data-setup dependencies already available")
}
+203
View File
@@ -0,0 +1,203 @@
#!/usr/bin/env Rscript
args <- commandArgs(trailingOnly = FALSE)
file_arg <- args[grepl("^--file=", args)][1]
if (!is.na(file_arg)) {
script_path <- sub("^--file=", "", file_arg)
repo_root <- normalizePath(file.path(dirname(script_path), "..", ".."), mustWork = FALSE)
} else {
repo_root <- normalizePath(getwd(), mustWork = TRUE)
}
if (!dir.exists(file.path(repo_root, "data-setup"))) repo_root <- normalizePath(getwd(), mustWork = TRUE)
raw_dir <- normalizePath(Sys.getenv("PARTY2D_RAW_DATA_DIR", file.path(repo_root, "_local", "raw")), mustWork = FALSE)
report_dir <- normalizePath(Sys.getenv("PARTY2D_REPORT_DIR", file.path(repo_root, "_local", "reports")), mustWork = FALSE)
dir.create(raw_dir, recursive = TRUE, showWarnings = FALSE)
dir.create(report_dir, recursive = TRUE, showWarnings = FALSE)
ua <- "party2d-data-setup/1.0 (+https://git.seimel.app/armin/party2d)"
download_file <- function(url, dest, overwrite = FALSE, headers = character()) {
dir.create(dirname(dest), recursive = TRUE, showWarnings = FALSE)
if (file.exists(dest) && file.info(dest)$size > 0 && !overwrite) {
message("OK existing: ", dest)
return(TRUE)
}
tmp <- paste0(dest, ".tmp")
if (file.exists(tmp)) unlink(tmp)
message("Downloading ", url, " -> ", dest)
ok <- tryCatch({
utils::download.file(
url,
tmp,
mode = "wb",
quiet = TRUE,
method = "libcurl",
headers = c("User-Agent" = ua, headers)
)
file.rename(tmp, dest)
}, error = function(e) {
message("FAILED ", url, ": ", conditionMessage(e))
FALSE
})
if (!ok && file.exists(tmp)) unlink(tmp)
isTRUE(ok)
}
download_dataverse <- function(doi, filename, dest, directory = NA_character_) {
if (!requireNamespace("jsonlite", quietly = TRUE)) {
message("FAILED Dataverse lookup: install R package jsonlite")
return(FALSE)
}
api <- paste0("https://dataverse.harvard.edu/api/datasets/:persistentId/?persistentId=", utils::URLencode(doi, reserved = TRUE))
fid <- tryCatch({
meta <- jsonlite::fromJSON(api, simplifyVector = FALSE)
files <- meta$data$latestVersion$files
matches <- Filter(function(x) {
same_file <- identical(x$dataFile$filename, filename)
same_dir <- is.na(directory) || identical(x$directoryLabel, directory)
same_file && same_dir
}, files)
if (length(matches) == 0) NA_integer_ else as.integer(matches[[1]]$dataFile$id)
}, error = function(e) {
message("FAILED Dataverse lookup ", doi, " ", filename, ": ", conditionMessage(e))
NA_integer_
})
if (is.na(fid)) {
message("FAILED Dataverse lookup ", doi, ": file not found: ", filename)
return(FALSE)
}
download_file(paste0("https://dataverse.harvard.edu/api/access/datafile/", fid), dest)
}
download_manifesto <- function(dest) {
key <- Sys.getenv("MANIFESTO_API_KEY", Sys.getenv("PARTY2D_MANIFESTO_API_KEY", ""))
if (!nzchar(key)) {
message("SKIP Manifesto: set MANIFESTO_API_KEY or PARTY2D_MANIFESTO_API_KEY")
return(FALSE)
}
url <- "https://manifesto-project.wzb.eu/api/v1/get_core?key=MPDS2025a&raw=true"
download_file(
url,
dest,
overwrite = TRUE,
headers = c("Referer" = "https://manifesto-project.wzb.eu/datasets", "API_KEY" = key)
)
}
download_vparty <- function(dest) {
email <- Sys.getenv("PARTY2D_VDEM_EMAIL", "")
if (!nzchar(email)) {
message("SKIP V-Party: set PARTY2D_VDEM_EMAIL to use V-Dem's required download form")
return(FALSE)
}
curl <- Sys.which("curl")
if (!nzchar(curl)) {
message("FAILED V-Party: curl is required for the provider form")
return(FALSE)
}
page <- "https://www.v-dem.net/data/v-party-dataset/country-party-date-v2/"
tmpdir <- tempfile("vparty")
dir.create(tmpdir)
on.exit(unlink(tmpdir, recursive = TRUE), add = TRUE)
cookie <- file.path(tmpdir, "cookies.txt")
html <- file.path(tmpdir, "page.html")
payload <- file.path(tmpdir, "payload.bin")
zip_path <- file.path(tmpdir, "vparty.zip")
status <- system2(curl, c("-L", "-A", ua, "-c", cookie, "-b", cookie, "-o", html, page), stdout = TRUE, stderr = TRUE)
if (!file.exists(html)) {
message("FAILED V-Party form load")
return(FALSE)
}
page_text <- paste(readLines(html, warn = FALSE), collapse = "\n")
csrf <- sub('.*name="csrfmiddlewaretoken"[^>]*value="([^"]*)".*', '\\1', page_text)
if (identical(csrf, page_text)) csrf <- ""
gender <- Sys.getenv("PARTY2D_VDEM_GENDER", "")
form <- c(
"csrfmiddlewaretoken", csrf,
"email", email,
"gender", gender,
"accept_terms", "on",
"dataset_file", "17",
"website", ""
)
args <- c(
"-L", "-A", ua, "-c", cookie, "-b", cookie,
"-H", paste0("Referer: ", page),
"-H", paste0("X-CSRFToken: ", csrf),
"-H", "X-Requested-With: XMLHttpRequest",
"-o", payload,
"--data-urlencode", paste0(form[1], "=", form[2]),
"--data-urlencode", paste0(form[3], "=", form[4]),
"--data-urlencode", paste0(form[5], "=", form[6]),
"--data-urlencode", paste0(form[7], "=", form[8]),
"--data-urlencode", paste0(form[9], "=", form[10]),
"--data-urlencode", paste0(form[11], "=", form[12]),
paste0(page, "#dataset-download")
)
system2(curl, args, stdout = TRUE, stderr = TRUE)
if (!file.exists(payload) || file.info(payload)$size == 0) {
message("FAILED V-Party form submit")
return(FALSE)
}
bytes <- readBin(payload, "raw", n = min(file.info(payload)$size, 4))
if (length(bytes) >= 2 && identical(as.integer(bytes[1:2]), c(0x50L, 0x4bL))) {
file.copy(payload, zip_path, overwrite = TRUE)
} else {
text <- paste(readLines(payload, warn = FALSE), collapse = "\n")
m <- regexpr('https?://[^" ]*CPD_V-Party_R_v2\\.zip|/[^" ]*CPD_V-Party_R_v2\\.zip', text)
if (m[1] < 0) {
message("FAILED V-Party form response did not include a recognizable ZIP download")
return(FALSE)
}
url <- regmatches(text, m)[1]
if (startsWith(url, "/")) url <- paste0("https://www.v-dem.net", url)
if (!download_file(url, zip_path, overwrite = TRUE, headers = c("Referer" = page))) return(FALSE)
}
listing <- utils::unzip(zip_path, list = TRUE)
member <- listing$Name[grepl("\\.(rds|rda|rdata)$", listing$Name, ignore.case = TRUE)][1]
if (is.na(member)) {
message("FAILED V-Party ZIP contains no R data file")
return(FALSE)
}
dir.create(dirname(dest), recursive = TRUE, showWarnings = FALSE)
utils::unzip(zip_path, files = member, exdir = tmpdir, overwrite = TRUE)
file.copy(file.path(tmpdir, member), dest, overwrite = TRUE)
message("Extracted V-Party R data: ", dest)
TRUE
}
status <- c()
status["PolDem"] <- download_file("https://poldem.eui.eu/downloads/cosa/poldem-election_all.csv", file.path(raw_dir, "poldem", "poldem-election_all.csv"))
status["PartyFacts external parties"] <- download_file("https://partyfacts.herokuapp.com/download/external-parties-csv/", file.path(raw_dir, "partyfacts", "partyfacts-external-parties.csv"))
status["Manifesto MPDS 2025a"] <- download_manifesto(file.path(raw_dir, "manifesto", "MPDataset_MPDS2025a.csv"))
ches_base <- "https://www.chesdata.eu/s/"
ches_files <- c(
"1999-2019_CHES_dataset_meansv3.csv" = "1999-2019_CHES_dataset_means(v3).csv",
"1999-2024_CHES_dataset_meansV2-3k4l.csv" = "1999-2024_CHES_dataset_meansV2-3k4l.csv",
"CHES_2024_final_v2.csv" = "CHES_2024_final_v2.csv",
"CHES_2024_ALL_Stacked_Expert.csv" = "CHES_2024_expert_level.csv",
"CHES_CA2023.csv" = "CHES_CA2023.csv",
"CHES_CA2023_expert-level.csv" = "CHES_CA2023_expert_level.csv",
"ches_la_2020_aggregate_level_v01.csv" = "ches_la_2020_aggregate_level_v01.csv",
"ches_la_2020_expert_level_v01.csv" = "CHES_LA2020_expert_level.csv",
"CHES_ISRAEL_means_2021_2022.csv" = "CHES_ISRAEL_means_2021_2022.csv",
"CHES_ISRAEL_expert_level_2021_2022.csv" = "CHES_IL_expert_level.csv"
)
for (remote in names(ches_files)) {
local <- ches_files[[remote]]
status[paste("CHES", local)] <- download_file(paste0(ches_base, utils::URLencode(remote, reserved = TRUE)), file.path(raw_dir, "ches", local))
}
status["POPPA integrated v2"] <- download_dataverse("doi:10.7910/DVN/RMQREQ", "poppa_integrated_v2.rds", file.path(raw_dir, "poppa", "poppa_integrated_v2.rds"), "final_data_v2")
status["GPS 2019 party"] <- download_dataverse("doi:10.7910/DVN/WMGTNS", "Global Party Survey by Party SPSS V2_1_Apr_2020-2.tab", file.path(raw_dir, "gps", "Global Party Survey by Party SPSS V2_1_Apr_2020-2.tab"))
status["V-Party"] <- download_vparty(file.path(raw_dir, "vparty", "V-Dem-CPD-Party-V2.rds"))
report <- file.path(report_dir, "download_sources_report.md")
lines <- c("# Download sources report", "", "| source | status |", "| --- | --- |")
for (name in names(status)) lines <- c(lines, paste0("| ", name, " | ", if (isTRUE(status[[name]])) "ok" else "missing/failed", " |"))
writeLines(lines, report)
message("Wrote download report: ", report)
quit(status = if (all(status)) 0 else 2)
+348
View File
@@ -0,0 +1,348 @@
# ============================================================
# 02_build_model_inputs.R - Master Data Pipeline Orchestrator
# ============================================================
# Coordinates all data processing sub-scripts and produces
# final output files for the two-dimensional party-position model. By default this writes only
# to local-only directories under _local/ and never overwrites committed data/.
#
# Sub-scripts (run conditionally based on intermediate file existence):
# process_manifesto.R -> manifesto_data.csv
# process_poldem.R -> poldem_data.csv
# process_expert.R -> expert_raw.csv, lr_data_raw.csv
# process_morgan.R -> morgan_data.csv, morgan_lr.csv
#
# Final generated model inputs:
# text_data.csv - Combined manifesto + PolDem
# expert.csv - Expert survey data (CHES, V-Party, POPPA, GPS)
# lr_data.csv - General left-right anchoring data
# ============================================================
library(tidyverse)
library(countrycode)
cmd_args <- commandArgs(trailingOnly = FALSE)
file_arg <- grep("^--file=", cmd_args, value = TRUE)
if (length(file_arg) > 0) {
this_file <- normalizePath(sub("^--file=", "", file_arg[[1]]), mustWork = TRUE)
repo_root <- normalizePath(file.path(dirname(this_file), "..", ".."), mustWork = TRUE)
} else {
repo_root <- normalizePath(getwd(), mustWork = TRUE)
}
script_dir <- file.path(repo_root, "data-setup", "R")
build_dir <- normalizePath(
Sys.getenv("PARTY2D_BUILD_DIR", file.path(repo_root, "_local", "build")),
mustWork = FALSE
)
generated_input_dir <- normalizePath(
Sys.getenv("PARTY2D_GENERATED_INPUT_DIR", file.path(repo_root, "_local", "generated-inputs")),
mustWork = FALSE
)
dir.create(build_dir, recursive = TRUE, showWarnings = FALSE)
dir.create(generated_input_dir, recursive = TRUE, showWarnings = FALSE)
# The source-processing scripts use relative paths for intermediate files.
# Keep those intermediates in the ignored build directory, never committed data/.
setwd(build_dir)
# Static model support inputs are versioned in data/ and copied into the local
# generated-input set for comparison. They are not regenerated by raw-source setup.
for (support_file in c("union_mapping.csv", "party_families.csv")) {
src <- file.path(repo_root, "data", support_file)
if (!file.exists(src)) {
stop("Required committed support input not found: ", src)
}
file.copy(src, file.path(build_dir, support_file), overwrite = TRUE)
}
cat("============================================================\n")
cat("Data Management Pipeline\n")
cat("============================================================\n\n")
cat("Build directory: ", build_dir, "\n", sep = "")
cat("Generated input directory: ", generated_input_dir, "\n\n", sep = "")
# ============================================================
# Configuration: Set to TRUE to force re-run of sub-scripts
# ============================================================
FORCE_RERUN_MANIFESTO <- FALSE
FORCE_RERUN_POLDEM <- FALSE
FORCE_RERUN_EXPERT <- FALSE
FORCE_RERUN_MORGAN <- FALSE
# ============================================================
# Step 1: Manifesto Data
# ============================================================
cat("Step 1: Manifesto data\n")
if (!file.exists("manifesto_data.csv") || !file.exists("election_data.csv") || FORCE_RERUN_MANIFESTO) {
cat(" Running process_manifesto.R...\n")
source(file.path(script_dir, "process_manifesto.R"))
} else {
cat(" Loading cached manifesto_data.csv and election_data.csv...\n")
}
manifesto <- read_csv("manifesto_data.csv", show_col_types = FALSE)
election_data <- read_csv("election_data.csv", show_col_types = FALSE)
cat(sprintf(" Loaded manifesto: %d rows, %d parties\n", nrow(manifesto), n_distinct(manifesto$party)))
cat(sprintf(" Loaded election: %d rows, %d parties\n\n", nrow(election_data), n_distinct(election_data$party)))
# ============================================================
# Step 2: PolDem Media Data
# ============================================================
cat("Step 2: PolDem media data\n")
if (!file.exists("poldem_data.csv") || FORCE_RERUN_POLDEM) {
cat(" Running process_poldem.R...\n")
source(file.path(script_dir, "process_poldem.R"))
} else {
cat(" Loading cached poldem_data.csv...\n")
}
poldem_data <- read_csv("poldem_data.csv", show_col_types = FALSE)
cat(sprintf(" Loaded: %d rows, %d parties\n\n", nrow(poldem_data), n_distinct(poldem_data$party)))
# ============================================================
# Step 4: Expert Survey Data
# ============================================================
cat("Step 3: Expert survey data\n")
if (!file.exists("expert_raw.csv") || !file.exists("lr_data_raw.csv") || FORCE_RERUN_EXPERT) {
cat(" Running process_expert.R...\n")
source(file.path(script_dir, "process_expert.R"))
} else {
cat(" Loading cached expert_raw.csv and lr_data_raw.csv...\n")
}
expert_raw <- read_csv("expert_raw.csv", show_col_types = FALSE)
lr_data_raw <- read_csv("lr_data_raw.csv", show_col_types = FALSE)
cat(sprintf(" Expert: %d rows, LR: %d rows\n\n", nrow(expert_raw), nrow(lr_data_raw)))
# ============================================================
# Step 3b: Morgan (1976) Historical Expert Data
# ============================================================
cat("Step 3b: Morgan (1976) historical L-R data\n")
# First run to generate morgan_data.csv if needed
if (!file.exists("morgan_data.csv") || FORCE_RERUN_MORGAN) {
cat(" Running process_morgan.R (initial processing)...\n")
source(file.path(script_dir, "process_morgan.R"))
}
# morgan_lr.csv depends on text_data.csv, so we need to check if it needs regeneration
# It will be generated/regenerated below after text_data is created
# ============================================================
# Step 4: Combine Text Data Sources
# ============================================================
cat("Step 4: Combining text data sources\n")
text_data <- bind_rows(manifesto, poldem_data)
cat(sprintf(" Combined text_data: %d rows\n", nrow(text_data)))
# Save unfiltered text_data for reproducible mismatch diagnosis
write_csv(text_data, "text_data_unfiltered.csv")
cat(sprintf(" Saved unfiltered text_data: %d rows, %d parties\n", nrow(text_data), n_distinct(text_data$party)))
# ============================================================
# Step 4b: Party Renames (applied before filtering)
# ============================================================
# Renames must happen BEFORE the relevance filter so that party IDs
# match across text_data and expert_raw when computing expert coverage.
# Simple renames only (organizational continuity: same leadership/members)
simple_renames <- c(
`10` = 1816L, # DE: Greens -> Bündnis90/Grüne
`276` = 120L, # RO: FDSN/PDSR -> PSD (renamed 2001)
`8054` = 878L, # IT: PDS -> DS (renamed 1998)
`1696` = 813L, # IT: MSI -> AN (refounded 1995)
`553` = 1968L, # BE: Vlaams Blok -> Vlaams Belang (refounded 2004)
`8058` = 1626L # IT: Forza Italia (refounded 2013) -> Forza Italia (same party, Berlusconi)
)
apply_simple_renames <- function(df) {
for (old_id in names(simple_renames)) {
df <- df %>%
mutate(party = ifelse(party == as.integer(old_id), simple_renames[[old_id]], party))
}
df
}
cat("\nStep 4b: Party renames\n")
text_data <- apply_simple_renames(text_data)
cat(sprintf(" Applied %d renames to text_data\n", length(simple_renames)))
# ============================================================
# Step 4c: Relevance Filter
# ============================================================
# Design: R pipeline filters for RELEVANCE (is this party worth modeling?).
# Julia pipeline handles INTERPOLATION QUALITY (MAX_GAP=7 segment splitting, MIN_OBS=2).
# Expert survey coverage is a relevance signal: CHES only covers parties with >1% vote share.
cat("\nStep 4c: Relevance filter\n")
parties_before <- n_distinct(text_data$party)
# Compute expert coverage per party (with renames applied for consistent matching)
expert_year_counts <- bind_rows(
expert_raw %>% select(party, year),
lr_data_raw %>% select(party, year)
) %>% distinct() %>%
apply_simple_renames() %>%
distinct() %>%
count(party, name = "expert_years")
expert_party_ids <- unique(expert_year_counts$party)
cat(sprintf(" Parties with expert data: %d\n", length(expert_party_ids)))
# Three-tier relevance filter:
# Tier 1: 3+ text data years (always include, regardless of expert data)
# Tier 2: 2 text years + any expert data (major newer parties like M5S, ANO, LREM)
# Tier 3: 1 text year + 3+ expert survey years (parties with rich expert coverage)
text_data <- text_data %>%
group_by(country, party) %>%
mutate(n_years = n_distinct(year)) %>%
ungroup() %>%
left_join(expert_year_counts, by = "party") %>%
mutate(expert_years = replace_na(expert_years, 0L)) %>%
mutate(
tier = case_when(
n_years >= 3 ~ 1L,
n_years >= 2 & party %in% expert_party_ids ~ 2L,
n_years >= 1 & expert_years >= 3 ~ 3L,
TRUE ~ 0L
)
) %>%
filter(tier > 0) %>%
select(-n_years, -expert_years, -tier)
parties_after <- n_distinct(text_data$party)
cat(sprintf(" Parties before filter: %d\n", parties_before))
cat(sprintf(" Parties after filter: %d\n", parties_after))
cat(sprintf(" Parties removed: %d\n\n", parties_before - parties_after))
# ============================================================
# Step 5: Party Harmonization
# ============================================================
cat("Step 5: Party harmonization (union-aware)\n")
# Load union mapping to identify constituent parties
union_map <- read_csv("union_mapping.csv", show_col_types = FALSE)
# Build set of constituent parties whose union is in text_data
constituent_parties <- union_map %>%
filter(manifesto_pf_id %in% unique(text_data$party)) %>%
pull(expert_pf_id)
cat(sprintf(" Union mappings loaded: %d rows covering %d unions\n",
nrow(union_map), n_distinct(union_map$manifesto_pf_id)))
cat(sprintf(" Constituent parties with unions in text_data: %d\n",
length(unique(constituent_parties))))
# Deduplicate union manifesto rows: where multiple CMP codes map to the same
# union PF ID with identical content, keep only one set per (party, year, var)
text_data_before_dedup <- nrow(text_data)
text_data <- text_data %>%
distinct(country, party, year, var, .keep_all = TRUE)
cat(sprintf(" Text data: %d unique parties after harmonization\n", n_distinct(text_data$party)))
cat(sprintf(" Text data: deduplicated %d -> %d rows\n", text_data_before_dedup, nrow(text_data)))
# Filter expert data: keep parties in text_data OR constituent parties of unions in text_data
expert <- expert_raw %>%
apply_simple_renames() %>%
group_by(country, party, var, year) %>%
summarise(
val = mean(val, na.rm = TRUE),
val_int = first(val_int),
n_scale = first(n_scale),
n_experts = first(n_experts),
project = first(project),
type_low = first(type_low),
type_high = first(type_high),
.groups = "drop"
) %>%
filter(party %in% unique(text_data$party) | party %in% constituent_parties)
lr_data <- lr_data_raw %>%
apply_simple_renames() %>%
group_by(country, party, var, year) %>%
summarise(
val = mean(val, na.rm = TRUE),
val_int = first(val_int),
n_scale = first(n_scale),
n_experts = first(n_experts),
project = first(project),
.groups = "drop"
) %>%
filter(party %in% unique(text_data$party) | party %in% constituent_parties)
cat(sprintf(" Expert data: %d rows (filtered to text_data parties)\n", nrow(expert)))
cat(sprintf(" LR data (CHES/POPPA): %d rows (filtered to text_data parties)\n", nrow(lr_data)))
# ============================================================
# Step 5b: Integrate Morgan L-R Data
# ============================================================
cat("\nStep 5b: Morgan L-R data integration\n")
# Generate morgan_lr.csv (requires text_data.csv to exist)
# We need to regenerate it if text_data changed or if forced
if (!file.exists("morgan_lr.csv") || FORCE_RERUN_MORGAN) {
cat(" Generating morgan_lr.csv...\n")
# Write text_data first so morgan script can use it
write_csv(text_data, "text_data.csv")
source(file.path(script_dir, "process_morgan.R"))
}
# Load and integrate Morgan L-R data
if (file.exists("morgan_lr.csv")) {
morgan_lr <- read_csv("morgan_lr.csv", show_col_types = FALSE) %>%
apply_simple_renames() %>%
filter(party %in% unique(text_data$party) | party %in% constituent_parties)
cat(sprintf(" Morgan L-R: %d rows (filtered to text_data parties)\n", nrow(morgan_lr)))
cat(sprintf(" Morgan parties: %d\n", n_distinct(morgan_lr$party)))
cat(sprintf(" Morgan year range: %d-%d\n", min(morgan_lr$year), max(morgan_lr$year)))
# Combine with existing lr_data
lr_data_before <- nrow(lr_data)
lr_data <- bind_rows(lr_data, morgan_lr) %>%
arrange(country, party, year, var)
cat(sprintf(" Combined LR data: %d rows (+%d from Morgan)\n",
nrow(lr_data), nrow(lr_data) - lr_data_before))
} else {
cat(" Warning: morgan_lr.csv not found, skipping Morgan integration\n")
}
cat("\n")
# ============================================================
# Step 6: Write Final Outputs
# ============================================================
cat("Step 6: Writing final outputs\n")
write_csv(text_data, "text_data.csv")
write_csv(expert, "expert.csv")
write_csv(lr_data, "lr_data.csv")
for (final_file in c("text_data.csv", "expert.csv", "lr_data.csv", "union_mapping.csv", "party_families.csv")) {
file.copy(file.path(build_dir, final_file), file.path(generated_input_dir, final_file), overwrite = TRUE)
}
cat("\n============================================================\n")
cat("Pipeline Complete!\n")
cat("============================================================\n\n")
cat("Output files written:\n")
cat(sprintf(" local generated input dir: %s\n", generated_input_dir))
cat(sprintf(" text_data.csv: %d rows\n", nrow(text_data)))
cat(sprintf(" - Manifesto: %d rows\n", sum(grepl("_manifesto", text_data$var))))
cat(sprintf(" - PolDem: %d rows\n", sum(grepl("_poldem", text_data$var))))
cat(sprintf(" expert.csv: %d rows\n", nrow(expert)))
cat(sprintf(" lr_data.csv: %d rows\n", nrow(lr_data)))
cat(sprintf(" - CHES: %d rows\n", sum(lr_data$var == "lr_ches")))
cat(sprintf(" - POPPA: %d rows\n", sum(lr_data$var == "lr_poppa")))
cat(sprintf(" - Morgan: %d rows\n", sum(lr_data$var == "lr_morgan")))
cat("\nUnique parties in text_data:", n_distinct(text_data$party), "\n")
cat("Countries:", paste(sort(unique(text_data$country)), collapse = ", "), "\n")
cat("Year range:", min(text_data$year, na.rm = TRUE), "-", max(text_data$year, na.rm = TRUE), "\n")
@@ -0,0 +1,53 @@
#!/usr/bin/env Rscript
args <- commandArgs(trailingOnly = FALSE)
file_arg <- args[grepl("^--file=", args)][1]
if (!is.na(file_arg)) {
script_path <- sub("^--file=", "", file_arg)
repo_root <- normalizePath(file.path(dirname(script_path), "..", ".."), mustWork = FALSE)
} else {
repo_root <- normalizePath(getwd(), mustWork = TRUE)
}
if (!dir.exists(file.path(repo_root, "data"))) repo_root <- normalizePath(getwd(), mustWork = TRUE)
generated_dir <- Sys.getenv("PARTY2D_GENERATED_INPUT_DIR", file.path(repo_root, "_local", "generated-inputs"))
report_dir <- Sys.getenv("PARTY2D_REPORT_DIR", file.path(repo_root, "_local", "reports"))
dir.create(report_dir, recursive = TRUE, showWarnings = FALSE)
files <- c("text_data.csv", "expert.csv", "lr_data.csv", "union_mapping.csv", "party_families.csv")
file_info <- lapply(files, function(file) {
committed <- file.path(repo_root, "data", file)
generated <- file.path(generated_dir, file)
committed_exists <- file.exists(committed)
generated_exists <- file.exists(generated)
committed_size <- if (committed_exists) file.info(committed)$size else NA_real_
generated_size <- if (generated_exists) file.info(generated)$size else NA_real_
identical_bytes <- committed_exists && generated_exists && isTRUE(tools::md5sum(committed) == tools::md5sum(generated))
data.frame(
file = file,
committed_exists = committed_exists,
generated_exists = generated_exists,
committed_size = committed_size,
generated_size = generated_size,
identical_bytes = identical_bytes,
stringsAsFactors = FALSE
)
})
summary <- do.call(rbind, file_info)
report <- file.path(report_dir, "input_comparison.md")
lines <- c(
"# Model input comparison",
"",
paste0("Generated: ", format(Sys.time(), "%Y-%m-%d %H:%M:%S %Z")),
"",
"| File | Committed exists | Generated exists | Committed bytes | Generated bytes | Identical bytes |",
"| --- | --- | --- | ---: | ---: | --- |",
apply(summary, 1, function(row) {
paste0("| ", row[["file"]], " | ", row[["committed_exists"]], " | ", row[["generated_exists"]], " | ", row[["committed_size"]], " | ", row[["generated_size"]], " | ", row[["identical_bytes"]], " |")
})
)
writeLines(lines, report)
print(summary, row.names = FALSE)
message("Comparison report written to ", report)
+620
View File
@@ -0,0 +1,620 @@
# ============================================================
# process_expert.R - Expert Survey Data Processing
# ============================================================
# Processes expert survey data from multiple sources:
# - Chapel Hill Expert Survey (CHES)
# - V-Party Dataset
# - POPPA
# - GPS (Norris)
#
# Outputs: expert_raw.csv, lr_data_raw.csv
#
# V5 changes:
# - val_int (integer rounded to nearest scale point) and n_scale columns
# - n_experts column preserved (not dropped)
# - V-Party cultural expansion: 5 native items replace GPS ep_v6_lib_cons
# - V-Party economic expansion: v2pawelf added
# - Reverse-coding for V-Party cultural + welfare items
# ============================================================
library(tidyverse)
library(countrycode)
library(haven)
library(foreign)
# Set working directory (works both in RStudio and command line)
if (interactive() && requireNamespace("rstudioapi", quietly = TRUE)) {
try(setwd(dirname(rstudioapi::getActiveDocumentContext()$path)), silent = TRUE)
}
cat("Processing expert survey data...\n")
raw_data_dir <- Sys.getenv(
"PARTY2D_RAW_DATA_DIR",
unset = file.path("..", "..", "_local", "raw")
)
ches_dir <- file.path(raw_data_dir, "ches")
vparty_dir <- file.path(raw_data_dir, "vparty")
poppa_dir <- file.path(raw_data_dir, "poppa")
gps_dir <- file.path(raw_data_dir, "gps")
partyfacts_path <- file.path(raw_data_dir, "partyfacts", "partyfacts-external-parties.csv")
ches_country_iso2 <- function(country_id) {
lookup <- c(
`1` = "BE", `2` = "DK", `3` = "DE", `4` = "GR", `5` = "ES",
`6` = "FR", `7` = "IE", `8` = "IT", `10` = "NL", `11` = "GB",
`12` = "PT", `13` = "AT", `14` = "FI", `16` = "SE", `20` = "BG",
`21` = "CZ", `22` = "EE", `23` = "HU", `24` = "LV", `25` = "LT",
`26` = "PL", `27` = "RO", `28` = "SK", `29` = "SI", `31` = "HR",
`32` = "TR", `33` = "NO", `34` = "CH", `35` = "MT", `36` = "CY",
`37` = "IS", `38` = "CH", `40` = "CY"
)
unname(lookup[as.character(country_id)])
}
ches2024_country_iso2 <- function(country_id) {
lookup <- c(
`1` = "BE", `2` = "DK", `3` = "DE", `4` = "GR", `5` = "ES",
`6` = "FR", `7` = "IE", `8` = "IT", `10` = "NL", `11` = "GB",
`12` = "PT", `13` = "AT", `14` = "FI", `16` = "SE", `20` = "BG",
`21` = "CZ", `22` = "EE", `23` = "HU", `24` = "LV", `25` = "LT",
`26` = "PL", `27` = "RO", `28` = "SK", `29` = "SI", `31` = "HR",
`34` = "TR", `35` = "NO", `36` = "CH", `37` = "MT", `40` = "CY",
`45` = "IS"
)
unname(lookup[as.character(country_id)])
}
# ============================================================
# PartyFacts Linkage for CHES
# ============================================================
partyfacts_raw <- read_csv(partyfacts_path, show_col_types = FALSE)
ches_link <- partyfacts_raw %>%
filter(dataset_key == "ches") %>%
transmute(id = dataset_party_id,
country = countrycode(country, origin = 'iso3c', destination = "iso2c"),
party = partyfacts_id)
# ============================================================
# Expert Count Tables (from individual response files)
# ============================================================
cat(" Loading expert count tables from individual response files...\n")
# CHES 2024: dual lookup (party_id primary, country+name fallback for ID mismatches)
ches24_exp_raw <- read_csv(file.path(ches_dir, 'CHES_2024_expert_level.csv'), show_col_types = FALSE)
ches24_exp_by_id <- ches24_exp_raw %>%
group_by(party_id) %>%
summarise(n_experts_id = as.integer(n_distinct(id)), .groups = "drop")
ches24_exp_by_name <- ches24_exp_raw %>%
mutate(country_iso2 = countrycode(cname, origin = "country.name", destination = "iso2c")) %>%
group_by(country_iso2, party_name) %>%
summarise(n_experts_name = as.integer(n_distinct(id)), .groups = "drop")
ches_ca_expert_counts <- read_csv(file.path(ches_dir, 'CHES_CA2023_expert_level.csv'), show_col_types = FALSE) %>%
group_by(party_id) %>%
summarise(n_experts = as.integer(n_distinct(expert)), .groups = "drop")
ches_la_expert_counts <- read_csv(file.path(ches_dir, 'CHES_LA2020_expert_level.csv'), show_col_types = FALSE) %>%
group_by(party_id) %>%
summarise(n_experts = as.integer(n_distinct(expert_id)), .groups = "drop")
ches_il_expert_counts <- read_csv(file.path(ches_dir, 'CHES_IL_expert_level.csv'), show_col_types = FALSE) %>%
group_by(party_id, year) %>%
summarise(n_experts = as.integer(n_distinct(id)), .groups = "drop")
# ============================================================
# Chapel Hill Expert Survey (CHES) - 1999-2019
# ============================================================
cat(" Processing CHES 1999-2019...\n")
ches <- read_csv(file.path(ches_dir, '1999-2019_CHES_dataset_means(v3).csv'), show_col_types = FALSE) %>%
rename(country_id = country) %>%
transmute(country = ches_country_iso2(country_id),
vote = vote,
year = year,
id = as.character(party_id),
project = 'CHES',
n_experts = as.integer(expert),
lrecon_ches = lrecon/10,
galtan_ches = galtan/10) %>%
pivot_longer(cols = lrecon_ches:galtan_ches, names_to = 'var', values_to = 'val') %>%
mutate(n_scale = 10L) %>%
left_join(ches_link, by = c("id", "country")) %>%
filter(!is.na(val), !is.na(party), !is.na(country)) %>%
select(-id) %>%
mutate(type_low = ifelse(var == "lrecon_ches", "pro_welfare", "cosmopolitan"),
type_high = ifelse(var == "lrecon_ches", "pro_market", "traditional"))
# ============================================================
# CHES 2024 Update
# ============================================================
cat(" Processing CHES 2024...\n")
# Country code lookup for CHES 2024 format
country_lookup <- c(
"be" = "BE", "dk" = "DK", "ge" = "DE", "gr" = "GR", "esp" = "ES",
"fr" = "FR", "irl" = "IE", "it" = "IT", "nl" = "NL", "uk" = "GB",
"por" = "PT", "aus" = "AT", "fin" = "FI", "sv" = "SE", "bul" = "BG",
"cz" = "CZ", "est" = "EE", "hun" = "HU", "lat" = "LV", "lith" = "LT",
"pol" = "PL", "rom" = "RO", "slo" = "SK", "sle" = "SI", "cro" = "HR",
"tur" = "TR", "nor" = "NO", "swi" = "CH", "mal" = "MT", "cyp" = "CY",
"ice" = "IS"
)
convert_country_codes <- function(codes) {
numeric_result <- ches2024_country_iso2(codes)
result <- country_lookup[codes]
result[is.na(result)] <- numeric_result[is.na(result)]
result[is.na(result)] <- codes[is.na(result)]
return(unname(result))
}
ches24 <- read_csv(file.path(ches_dir, 'CHES_2024_final_v2.csv'), show_col_types = FALSE) %>%
mutate(country_iso2 = convert_country_codes(country)) %>%
left_join(ches24_exp_by_id, by = "party_id") %>%
left_join(ches24_exp_by_name, by = c("country_iso2", "party" = "party_name")) %>%
transmute(country = country_iso2,
vote = vote,
year = 2024,
id = as.character(party_id),
project = 'CHES',
n_experts = coalesce(n_experts_id, n_experts_name),
lrecon_ches = lrecon/10,
galtan_ches = galtan/10) %>%
pivot_longer(cols = lrecon_ches:galtan_ches, names_to = 'var', values_to = 'val') %>%
mutate(n_scale = 10L) %>%
left_join(ches_link, by = c("id", "country")) %>%
filter(!is.na(val), !is.na(party), !is.na(country)) %>%
select(-id) %>%
mutate(type_low = ifelse(var == "lrecon_ches", "pro_welfare", "cosmopolitan"),
type_high = ifelse(var == "lrecon_ches", "pro_market", "traditional"))
ches <- bind_rows(ches, ches24)
# ============================================================
# CHES Canada 2023
# ============================================================
cat(" Processing CHES Canada 2023...\n")
ches_ca <- read_csv(file.path(ches_dir, 'CHES_CA2023.csv'), show_col_types = FALSE) %>%
filter(!is.na(partyfacts_id)) %>%
left_join(ches_ca_expert_counts, by = "party_id") %>%
transmute(country = "CA",
year = 2023,
party = partyfacts_id,
project = 'CHES',
n_experts = n_experts,
lrecon_ches = lrecon/10,
galtan_ches = galtan/10) %>%
pivot_longer(cols = lrecon_ches:galtan_ches, names_to = 'var', values_to = 'val') %>%
mutate(n_scale = 10L) %>%
filter(!is.na(val), !is.na(party)) %>%
mutate(type_low = ifelse(var == "lrecon_ches", "pro_welfare", "cosmopolitan"),
type_high = ifelse(var == "lrecon_ches", "pro_market", "traditional"))
ches <- bind_rows(ches, ches_ca)
cat(sprintf(" CHES Canada: %d observations\n", nrow(ches_ca)))
# ============================================================
# CHES Latin America 2020
# ============================================================
cat(" Processing CHES Latin America 2020...\n")
ches_la_link <- partyfacts_raw %>%
filter(dataset_key == "ches") %>%
transmute(id = as.character(dataset_party_id),
country = countrycode(country, origin = "iso3c", destination = "iso2c"),
party = partyfacts_id)
ches_la <- read_csv(file.path(ches_dir, 'ches_la_2020_aggregate_level_v01.csv'), show_col_types = FALSE) %>%
left_join(ches_la_expert_counts, by = "party_id") %>%
transmute(country = toupper(country_abb),
year = 2020,
id = as.character(party_id),
project = 'CHES',
n_experts = n_experts,
lrecon_ches = lrecon/10,
galtan_ches = galtan/10) %>%
pivot_longer(cols = lrecon_ches:galtan_ches, names_to = 'var', values_to = 'val') %>%
mutate(n_scale = 10L) %>%
left_join(ches_la_link, by = c("id", "country")) %>%
filter(!is.na(val), !is.na(party), !is.na(country)) %>%
select(-id) %>%
mutate(type_low = as.character(ifelse(var == "lrecon_ches", "pro_welfare", "cosmopolitan")),
type_high = as.character(ifelse(var == "lrecon_ches", "pro_market", "traditional")))
ches <- bind_rows(ches, ches_la)
cat(sprintf(" CHES Latin America: %d observations\n", nrow(ches_la)))
# ============================================================
# CHES Israel 2021-2022
# ============================================================
cat(" Processing CHES Israel 2021-2022...\n")
ches_il_link <- partyfacts_raw %>%
filter(dataset_key == "ches") %>%
transmute(id = as.character(dataset_party_id),
country = countrycode(country, origin = "iso3c", destination = "iso2c"),
party = partyfacts_id)
ches_il <- read_csv(file.path(ches_dir, 'CHES_ISRAEL_means_2021_2022.csv'), show_col_types = FALSE) %>%
left_join(ches_il_expert_counts, by = c("party_id", "year")) %>%
transmute(country = "IL",
year = year,
id = as.character(party_id),
project = 'CHES',
n_experts = n_experts,
lrecon_ches = lrecon/10,
galtan_ches = galtan/10) %>%
pivot_longer(cols = lrecon_ches:galtan_ches, names_to = 'var', values_to = 'val') %>%
mutate(n_scale = 10L) %>%
left_join(ches_il_link, by = c("id", "country")) %>%
filter(!is.na(val), !is.na(party), !is.na(country)) %>%
select(-id) %>%
mutate(type_low = ifelse(var == "lrecon_ches", "pro_welfare", "cosmopolitan"),
type_high = ifelse(var == "lrecon_ches", "pro_market", "traditional"))
ches <- bind_rows(ches, ches_il)
cat(sprintf(" CHES Israel: %d observations\n", nrow(ches_il)))
cat(sprintf(" CHES total: %d observations\n", nrow(ches)))
# ============================================================
# CHES General Left-Right (for anchoring)
# ============================================================
cat(" Processing CHES LR anchoring data...\n")
ches_lr <- read_csv(file.path(ches_dir, '1999-2019_CHES_dataset_means(v3).csv'), show_col_types = FALSE) %>%
rename(country_id = country) %>%
transmute(country = ches_country_iso2(country_id),
vote = vote,
year = year,
id = as.character(party_id),
project = 'CHES',
n_experts = as.integer(expert),
val = lrgen/10,
var = 'lr_ches',
n_scale = 10L) %>%
left_join(ches_link, by = c("id", "country")) %>%
filter(!is.na(val), !is.na(party), !is.na(country)) %>%
select(-id)
ches24_lr <- read_csv(file.path(ches_dir, 'CHES_2024_final_v2.csv'), show_col_types = FALSE) %>%
mutate(country_iso2 = convert_country_codes(country)) %>%
left_join(ches24_exp_by_id, by = "party_id") %>%
left_join(ches24_exp_by_name, by = c("country_iso2", "party" = "party_name")) %>%
transmute(country = country_iso2,
vote = vote,
year = 2024,
id = as.character(party_id),
project = 'CHES',
n_experts = coalesce(n_experts_id, n_experts_name),
val = lrgen/10,
var = 'lr_ches',
n_scale = 10L) %>%
left_join(ches_link, by = c("id", "country")) %>%
filter(!is.na(val), !is.na(party), !is.na(country)) %>%
select(-id)
# CHES Canada LR
ches_ca_lr <- read_csv(file.path(ches_dir, 'CHES_CA2023.csv'), show_col_types = FALSE) %>%
filter(!is.na(partyfacts_id)) %>%
left_join(ches_ca_expert_counts, by = "party_id") %>%
transmute(country = "CA",
year = 2023,
party = partyfacts_id,
project = 'CHES',
n_experts = n_experts,
val = lrgen/10,
var = 'lr_ches',
n_scale = 10L) %>%
filter(!is.na(val), !is.na(party))
# CHES Latin America LR
ches_la_lr <- read_csv(file.path(ches_dir, 'ches_la_2020_aggregate_level_v01.csv'), show_col_types = FALSE) %>%
left_join(ches_la_expert_counts, by = "party_id") %>%
transmute(country = toupper(country_abb),
year = 2020,
id = as.character(party_id),
project = 'CHES',
n_experts = n_experts,
val = lrgen/10,
var = 'lr_ches',
n_scale = 10L) %>%
left_join(ches_la_link, by = c("id", "country")) %>%
filter(!is.na(val), !is.na(party), !is.na(country)) %>%
select(-id)
# CHES Israel LR
ches_il_lr <- read_csv(file.path(ches_dir, 'CHES_ISRAEL_means_2021_2022.csv'), show_col_types = FALSE) %>%
left_join(ches_il_expert_counts, by = c("party_id", "year")) %>%
transmute(country = "IL",
year = year,
id = as.character(party_id),
project = 'CHES',
n_experts = n_experts,
val = lrgen/10,
var = 'lr_ches',
n_scale = 10L) %>%
left_join(ches_il_link, by = c("id", "country")) %>%
filter(!is.na(val), !is.na(party), !is.na(country)) %>%
select(-id)
ches_lr <- bind_rows(ches_lr, ches24_lr, ches_ca_lr, ches_la_lr, ches_il_lr)
# ============================================================
# V-Party Dataset (V5: expanded to 7 variables)
# ============================================================
cat(" Processing V-Party...\n")
vparty_raw <- readRDS(file.path(vparty_dir, 'V-Dem-CPD-Party-V2.rds'))
# Economic 1: v2pariglef_osp (0-6 scale, higher = more right, NO reverse)
vparty_econ1 <- vparty_raw %>%
transmute(
country = countrycode(country_name, origin = "country.name", destination = "iso2c"),
year = year,
party = pf_party_id,
project = "V-Party",
n_experts = as.integer(v2pariglef_nr),
val = v2pariglef_osp / 6,
val_int = as.integer(round(v2pariglef_osp)),
n_scale = 6L,
var = "lrecon_vparty",
type_low = "pro_welfare",
type_high = "pro_market"
) %>%
na.omit()
# Economic 2 (NEW): v2pawelf_osp (0-5 scale, higher = more welfare = LEFT, REVERSE)
vparty_econ2 <- vparty_raw %>%
transmute(
country = countrycode(country_name, origin = "country.name", destination = "iso2c"),
year = year,
party = pf_party_id,
project = "V-Party",
n_experts = as.integer(v2pawelf_nr),
val = 1 - v2pawelf_osp / 5,
val_int = 5L - as.integer(round(v2pawelf_osp)),
n_scale = 5L,
var = "welf_vparty",
type_low = "pro_welfare",
type_high = "pro_market"
) %>%
na.omit()
# Cultural 1 (NEW): v2paimmig_osp (0-4 scale, higher = more pro-immigration = GAL, REVERSE)
vparty_cult1 <- vparty_raw %>%
transmute(
country = countrycode(country_name, origin = "country.name", destination = "iso2c"),
year = year,
party = pf_party_id,
project = "V-Party",
n_experts = as.integer(v2paimmig_nr),
val = 1 - v2paimmig_osp / 4,
val_int = 4L - as.integer(round(v2paimmig_osp)),
n_scale = 4L,
var = "immig_vparty",
type_low = "cosmopolitan",
type_high = "traditional"
) %>%
na.omit()
# Cultural 2 (NEW): v2palgbt_osp (0-4 scale, higher = more pro-LGBT = GAL, REVERSE)
vparty_cult2 <- vparty_raw %>%
transmute(
country = countrycode(country_name, origin = "country.name", destination = "iso2c"),
year = year,
party = pf_party_id,
project = "V-Party",
n_experts = as.integer(v2palgbt_nr),
val = 1 - v2palgbt_osp / 4,
val_int = 4L - as.integer(round(v2palgbt_osp)),
n_scale = 4L,
var = "lgbt_vparty",
type_low = "cosmopolitan",
type_high = "traditional"
) %>%
na.omit()
# Cultural 3 (NEW): v2paculsup_osp (0-4 scale, higher = less cultural superiority = GAL, REVERSE)
vparty_cult3 <- vparty_raw %>%
transmute(
country = countrycode(country_name, origin = "country.name", destination = "iso2c"),
year = year,
party = pf_party_id,
project = "V-Party",
n_experts = as.integer(v2paculsup_nr),
val = 1 - v2paculsup_osp / 4,
val_int = 4L - as.integer(round(v2paculsup_osp)),
n_scale = 4L,
var = "culsup_vparty",
type_low = "cosmopolitan",
type_high = "traditional"
) %>%
na.omit()
# Cultural 4 (NEW): v2parelig_osp (0-4 scale, higher = less religious = GAL, REVERSE)
vparty_cult4 <- vparty_raw %>%
transmute(
country = countrycode(country_name, origin = "country.name", destination = "iso2c"),
year = year,
party = pf_party_id,
project = "V-Party",
n_experts = as.integer(v2parelig_nr),
val = 1 - v2parelig_osp / 4,
val_int = 4L - as.integer(round(v2parelig_osp)),
n_scale = 4L,
var = "relig_vparty",
type_low = "cosmopolitan",
type_high = "traditional"
) %>%
na.omit()
# Cultural 5 (NEW): v2pagender_osp (0-4 scale, higher = more pro-gender equality = GAL, REVERSE)
vparty_cult5 <- vparty_raw %>%
transmute(
country = countrycode(country_name, origin = "country.name", destination = "iso2c"),
year = year,
party = pf_party_id,
project = "V-Party",
n_experts = as.integer(v2pagender_nr),
val = 1 - v2pagender_osp / 4,
val_int = 4L - as.integer(round(v2pagender_osp)),
n_scale = 4L,
var = "gender_vparty",
type_low = "cosmopolitan",
type_high = "traditional"
) %>%
na.omit()
vparty <- bind_rows(vparty_econ1, vparty_econ2,
vparty_cult1, vparty_cult2, vparty_cult3,
vparty_cult4, vparty_cult5)
cat(sprintf(" V-Party: %d observations (7 variables)\n", nrow(vparty)))
cat(sprintf(" lrecon: %d, welf: %d\n", nrow(vparty_econ1), nrow(vparty_econ2)))
cat(sprintf(" immig: %d, lgbt: %d, culsup: %d, relig: %d, gender: %d\n",
nrow(vparty_cult1), nrow(vparty_cult2), nrow(vparty_cult3),
nrow(vparty_cult4), nrow(vparty_cult5)))
# ============================================================
# POPPA Dataset
# ============================================================
cat(" Processing POPPA...\n")
poppa <- readRDS(file.path(poppa_dir, 'poppa_integrated_v2.rds')) %>%
transmute(country = countrycode(country_short, origin = "iso3c", destination = "iso2c"),
party = partyfacts_id,
val = lrecon/10,
var = "lrecon_poppa",
type_low = "pro_welfare",
type_high = "pro_market",
n_experts = as.integer(n_experts),
n_scale = 10L,
year = as.numeric(sub(".*-\\s*(\\d+)", "\\1", wave)),
project = "POPPA") %>%
na.omit()
cat(sprintf(" POPPA: %d observations\n", nrow(poppa)))
# POPPA General LR
poppa_lr <- readRDS(file.path(poppa_dir, 'poppa_integrated_v2.rds')) %>%
transmute(country = countrycode(country_short, origin = "iso3c", destination = "iso2c"),
party = partyfacts_id,
val = lroverall/10,
var = "lr_poppa",
n_experts = as.integer(n_experts),
n_scale = 10L,
year = as.numeric(sub(".*-\\s*(\\d+)", "\\1", wave)),
project = "POPPA") %>%
na.omit()
# ============================================================
# GPS (Norris) Survey
# ============================================================
cat(" Processing GPS...\n")
gps <- read.delim(file.path(gps_dir, "Global Party Survey by Party SPSS V2_1_Apr_2020-2.tab")) %>%
transmute(n_experts = as.integer(Experts),
lrecon_gps = as.numeric(V4_Scale)/10,
libcon_gps = as.numeric(V6_Scale)/10,
party = ID_PartyFacts,
country = countrycode(ifelse(ISO == "MAC", "MKD", ISO), origin = "iso3c", destination = "iso2c"),
year = 2019,
n_scale = 10L,
project = "GPS") %>%
pivot_longer(cols = lrecon_gps:libcon_gps, names_to = 'var', values_to = 'val') %>%
mutate(type_low = ifelse(var == "lrecon_gps", "pro_welfare", "cosmopolitan"),
type_high = ifelse(var == "lrecon_gps", "pro_market", "traditional")) %>%
na.omit()
cat(sprintf(" GPS: %d observations\n", nrow(gps)))
# ============================================================
# Combine Expert Data
# ============================================================
cat(" Combining expert surveys...\n")
expert_raw <- select(ches, -vote) %>%
bind_rows(vparty) %>%
bind_rows(gps) %>%
bind_rows(poppa) %>%
unique() %>%
arrange(country, party, year, var) %>%
filter(!is.na(val), !is.na(party), !is.na(country), !is.na(var))
# Compute val_int for datasets that don't have it pre-computed
# V-Party already has val_int; CHES/GPS/POPPA need it computed from val * n_scale
expert_raw <- expert_raw %>%
mutate(
val_int = ifelse(is.na(val_int), as.integer(round(val * n_scale)), val_int),
val_int = pmin(pmax(val_int, 0L), n_scale)
)
# Boundary adjustments for continuous val (avoid exact 0 or 1 for Stan prior means)
expert_raw <- expert_raw %>%
mutate(
val = case_when(
val == 0 ~ val + 1e-4,
val == 1 ~ val - 1e-4,
TRUE ~ val
))
# ============================================================
# Combine LR Data
# ============================================================
lr_data_raw <- ches_lr %>%
bind_rows(poppa_lr) %>%
select(-any_of("vote"))
# Boundary adjustments for continuous val (avoid exact 0 or 1)
lr_data_raw <- lr_data_raw %>%
mutate(
val = case_when(
val == 0 ~ val + 1e-4,
val == 1 ~ val - 1e-4,
TRUE ~ val
))
# Compute val_int for LR data
lr_data_raw <- lr_data_raw %>%
mutate(
val_int = as.integer(round(val * n_scale)),
val_int = pmin(pmax(val_int, 0L), n_scale)
)
# ============================================================
# Write Outputs
# ============================================================
write_csv(expert_raw, "expert_raw.csv")
write_csv(lr_data_raw, "lr_data_raw.csv")
cat(sprintf("\nOutputs written:\n"))
cat(sprintf(" expert_raw.csv: %d rows\n", nrow(expert_raw)))
cat(sprintf(" lr_data_raw.csv: %d rows\n", nrow(lr_data_raw)))
cat("\n Expert data by source:\n")
expert_raw %>%
group_by(project) %>%
summarise(n = n(), .groups = "drop") %>%
print()
cat("\n New columns check:\n")
cat(sprintf(" val_int range: %d - %d\n", min(expert_raw$val_int), max(expert_raw$val_int)))
cat(sprintf(" n_scale values: %s\n", paste(sort(unique(expert_raw$n_scale)), collapse = ", ")))
cat(sprintf(" n_experts non-NA: %d / %d\n", sum(!is.na(expert_raw$n_experts)), nrow(expert_raw)))
+179
View File
@@ -0,0 +1,179 @@
# ============================================================
# process_manifesto.R - Manifesto Project Data Processing
# ============================================================
# Processes Manifesto Project data for the two-dimensional party-position model
# Input: $PARTY2D_RAW_DATA_DIR/manifesto/MPDataset_MPDS2025a.csv
# Output: manifesto_data.csv
# ============================================================
library(tidyverse)
library(countrycode)
library(purrr)
# Set working directory (works both in RStudio and command line)
if (interactive() && requireNamespace("rstudioapi", quietly = TRUE)) {
try(setwd(dirname(rstudioapi::getActiveDocumentContext()$path)), silent = TRUE)
}
cat("Processing Manifesto Project data...\n")
raw_data_dir <- Sys.getenv(
"PARTY2D_RAW_DATA_DIR",
unset = file.path("..", "..", "_local", "raw")
)
manifesto_raw_path <- file.path(raw_data_dir, "manifesto", "MPDataset_MPDS2025a.csv")
partyfacts_path <- file.path(raw_data_dir, "partyfacts", "partyfacts-external-parties.csv")
# ============================================================
# PartyFacts Linkage
# ============================================================
partyfacts_raw <- read_csv(partyfacts_path, show_col_types = FALSE)
manifesto_link <- partyfacts_raw %>%
filter(dataset_key == "manifesto") %>%
transmute(id = dataset_party_id,
country = countrycode(country, origin = 'iso3c', destination = "iso2c"),
party = partyfacts_id,
party = ifelse(party == 622, 604, party))
# ============================================================
# Load Manifesto Data
# ============================================================
manifesto_data <- read_csv(manifesto_raw_path, show_col_types = FALSE)
# ============================================================
# CMP Code Mapping to 4 Dimensions
# ============================================================
vars <- tribble(
~type, ~subtype, ~per_var, ~stance, ~label,
# pro_market
"pro_market", "Market Regulation", "per401", "Positive", "Free Market Economy",
"pro_market", "Economic Liberalization","per402", "Positive", "Incentives: Positive",
"pro_market", "Market Regulation", "per407", "Positive", "Protectionism: Negative",
"pro_market", "Economic Liberalization","per414", "Positive", "Economic Orthodoxy",
"pro_market", "Economic Liberalization","per505", "Positive", "Welfare State Limitation",
"pro_market", "Economic Liberalization","per507", "Positive", "Education Limitation",
"pro_market", "Economic Liberalization","per702", "Positive", "Labour Groups: Negative",
"pro_market", "Market Regulation", "per406", "Negative", "Protectionism: Positive",
"pro_market", "Market Regulation", "per412", "Negative", "Controlled Economy",
"pro_market", "Economic Liberalization","per504", "Negative", "Welfare State Expansion",
# pro_welfare
"pro_welfare", "Economic Intervention", "per403", "Positive", "Market Regulation",
"pro_welfare", "Economic Intervention", "per404", "Positive", "Economic Planning",
"pro_welfare", "Economic Intervention", "per412", "Positive", "Controlled Economy",
"pro_welfare", "Economic Intervention", "per413", "Positive", "Nationalisation",
"pro_welfare", "Social Services", "per504", "Positive", "Welfare State Expansion",
"pro_welfare", "Social Services", "per506", "Positive", "Education Expansion",
"pro_welfare", "Economic Intervention", "per701", "Positive", "Labour Groups: Positive",
"pro_welfare", "Economic Intervention", "per401", "Negative", "Free Market Economy",
"pro_welfare", "Social Services", "per505", "Negative", "Welfare State Limitation",
# cosmopolitan
"cosmopolitan", "Internationalism", "per107", "Positive", "Internationalism: Positive",
"cosmopolitan", "Internationalism", "per108", "Positive", "European Community/Union: Positive",
"cosmopolitan", "Multiculturalism", "per607", "Positive", "Multiculturalism: Positive",
"cosmopolitan", "Multiculturalism", "per201", "Positive", "Freedom and Human Rights",
"cosmopolitan", "Multiculturalism", "per604", "Positive", "traditional Morality: Negative",
"cosmopolitan", "Internationalism", "per109", "Negative", "Internationalism: Negative",
"cosmopolitan", "Multiculturalism", "per601", "Negative", "National Way of Life: Positive",
# traditional
"traditional", "National Identity", "per109", "Positive", "Internationalism: Negative",
"traditional", "Conservative Morality", "per110", "Positive", "European Community/Union: Negative",
"traditional", "National Identity", "per601", "Positive", "National Way of Life: Positive",
"traditional", "Conservative Morality", "per603", "Positive", "traditional Morality: Positive",
"traditional", "Conservative Morality", "per608", "Positive", "Multiculturalism: Negative",
"traditional", "Conservative Morality", "per605", "Positive", "Law and Order: Positive",
"traditional", "National Identity", "per107", "Negative", "Internationalism: Positive",
"traditional", "Conservative Morality", "per607", "Negative", "Multiculturalism: Positive"
)
# ============================================================
# Process Manifesto Data
# ============================================================
manifesto <- vars %>%
pmap_dfr(~ manifesto_data %>%
transmute(country = countrycode(countryname, origin = 'country.name', destination = 'iso2c'),
year = as.numeric(format(as.Date(edate, format = "%d/%m/%Y"), "%Y")),
id = as.character(party),
count = round(.data[[..3]]),
var = ..3,
label = ..5,
type = ..1,
subtype = ..2,
stance = ..4,
project = 'Manifesto Project') %>%
left_join(manifesto_link, by = c("id", "country")) %>%
select(-id)) %>%
group_by(party, country, year, subtype) %>%
summarise(
positive = sum(count[stance == "Positive"], na.rm = TRUE),
sample = sum(count, na.rm = TRUE),
type = first(type),
project = first(project),
.groups = "drop"
) %>%
na.omit() %>%
rename(var = subtype) %>%
# Convert to bipolar bridge structure (type_high/type_low)
mutate(
type_high = case_when(
type == "pro_welfare" ~ "pro_welfare",
type == "pro_market" ~ "pro_market",
type == "cosmopolitan" ~ "cosmopolitan",
type == "traditional" ~ "traditional"
),
type_low = case_when(
type %in% c("pro_welfare", "pro_market") ~ ifelse(type == "pro_welfare", "pro_market", "pro_welfare"),
type %in% c("cosmopolitan", "traditional") ~ ifelse(type == "cosmopolitan", "traditional", "cosmopolitan")
)
) %>%
select(-type) %>%
# Add _manifesto suffix to variable names
mutate(var = paste0(tolower(gsub(" ", "_", var)), "_manifesto"))
# ============================================================
# NOTE: Temporal continuity filter moved to the data setup orchestrator.
# This allows exempting parties that appear in parliamentary data
# (parties in parliament are by definition not fringe parties)
# ============================================================
cat("Skipping temporal filter (applied in 02_build_model_inputs.R after combining with other text data)\n")
cat(sprintf(" Parties: %d\n", n_distinct(manifesto$party)))
# ============================================================
# Write Output
# ============================================================
write_csv(manifesto, "manifesto_data.csv")
cat(sprintf("Output: manifesto_data.csv (%d rows, %d parties)\n",
nrow(manifesto), n_distinct(manifesto$party)))
# ============================================================
# Election Data Extraction (vote shares)
# ============================================================
cat("\nExtracting election data (pervote)...\n")
election_data <- manifesto_data %>%
transmute(
country = countrycode(countryname, origin = 'country.name', destination = 'iso2c'),
year = as.numeric(format(as.Date(edate, format = "%d/%m/%Y"), "%Y")),
id = as.character(party),
pervote = pervote
) %>%
left_join(manifesto_link, by = c("id", "country")) %>%
select(-id) %>%
filter(!is.na(party), !is.na(pervote)) %>%
# Keep one row per (party, country, year) — take max pervote if duplicates
group_by(party, country, year) %>%
summarise(pervote = max(pervote, na.rm = TRUE), .groups = "drop") %>%
arrange(country, party, year)
write_csv(election_data, "election_data.csv")
cat(sprintf("Output: election_data.csv (%d rows, %d parties)\n",
nrow(election_data), n_distinct(election_data$party)))
# Export manifesto_link for use by other scripts
# (poldem also needs it for CMP linkage)
+399
View File
@@ -0,0 +1,399 @@
# process_morgan.R
# Process Morgan (1976) expert party position data
#
# Source: Morgan, Michael-John (1976). "The Modelling of Governmental
# Coalition Formation: A Policy-Based Approach with Interval Measurement."
# PhD dissertation, University of Michigan.
#
# Data extracted from Appendix B.3 (Tables B.3.1-B.3.12) via OCR.
# Position scores are 25%-truncated means (midmeans) from expert surveys.
# Scale: 0-100 (left-right)
library(tidyverse)
cat("Processing Morgan (1976) expert party position data...\n")
raw_data_dir <- Sys.getenv(
"PARTY2D_RAW_DATA_DIR",
unset = file.path("..", "..", "_local", "raw")
)
morgan_raw_path <- file.path(raw_data_dir, "morgan", "morgan_positions_raw.csv")
partyfacts_path <- file.path(raw_data_dir, "partyfacts", "partyfacts-external-parties.csv")
# Load raw extracted data
morgan_raw <- read_csv(morgan_raw_path, show_col_types = FALSE)
cat(sprintf("Loaded %d party-period observations from %d countries\n",
nrow(morgan_raw), n_distinct(morgan_raw$country)))
# Load PartyFacts linkage data
partyfacts <- read_csv(partyfacts_path, show_col_types = FALSE)
# Filter to Morgan dataset entries
morgan_pf <- partyfacts %>%
filter(dataset_key == "morgan") %>%
select(country, name_short, name_english, year_first, year_last,
external_id, partyfacts_id) %>%
rename(party_abbrev_pf = name_short)
cat(sprintf("Found %d Morgan parties in PartyFacts\n", nrow(morgan_pf)))
# Map extracted abbreviations to PartyFacts abbreviations
# Some adjustments needed due to OCR/transcription differences
abbrev_map <- tribble(
~country, ~party_abbrev, ~party_abbrev_pf,
# Denmark
"DNK", "SOCd", "SOCD",
"DNK", "SOCL", "SOCL",
"DNK", "COMM", "COMM",
"DNK", "RAD", "RAD",
"DNK", "LIB", "LIB",
"DNK", "CONS", "CONS",
"DNK", "LS", "LS",
"DNK", "LC", "LC",
"DNK", "JUST", "JUST",
# Finland
"FIN", "SKDL", "SKDL",
"FIN", "SOCd", "SOCD",
"FIN", "PROG", "PROG",
"FIN", "AGR", "AGR",
"FIN", "SWPP", "SWPP",
"FIN", "CONS", "CONS",
"FIN", "NPF", "NPF",
"FIN", "PDEM", "PDEM",
"FIN", "SDWS", "SDWS",
"FIN", "CENT", "CENT",
"FIN", "FRP", "FRP",
"FIN", "LIB", "LIB",
# Iceland
"ISL", "COMM", "COMM",
"ISL", "SOCd", "SOCD",
"ISL", "PROG", "PROG",
"ISL", "LIB", "LIB",
"ISL", "INDP", "INDP",
"ISL", "CONS", "CONS",
"ISL", "LLIB", "LLIB",
# Norway
"NOR", "LAB", "LAB",
"NOR", "LIB", "LIB",
"NOR", "AGR", "AGR",
"NOR", "CONS", "CONS",
"NOR", "COMM", "COMM",
"NOR", "SOCL", "SOCL",
"NOR", "CHPP", "CHPP",
"NOR", "CENT", "CENT",
# Sweden
"SWE", "COMM", "COMM",
"SWE", "SOCd", "SOCD",
"SWE", "AGR", "AGR",
"SWE", "LIB", "LIB",
"SWE", "CONS", "CONS",
"SWE", "CENT", "CENT",
# Netherlands
"NLD", "CPN", "CPN",
"NLD", "SOCd", "SOCD",
"NLD", "RAD", "RAD",
"NLD", "KVP", "KVP",
"NLD", "CHU", "CHU",
"NLD", "LIB", "LIB",
"NLD", "ARP", "ARP",
"NLD", "SGP", "SGP",
"NLD", "NSB", "NSB",
"NLD", "PVDA", "PVDA",
"NLD", "VVD", "VVD",
"NLD", "PSP", "PSP",
"NLD", "PPR", "PPR",
"NLD", "D66", "D66",
"NLD", "DS70", "DS70",
"NLD", "GPV", "GPV",
"NLD", "BP", "BP",
# Belgium
"BEL", "COMM", "COMM",
"BEL", "POB", "POB",
"BEL", "CATH", "CATH",
"BEL", "LIB", "LIB",
"BEL", "FNAT", "FNAT",
"BEL", "REX", "REX",
"BEL", "PSB", "PSB",
"BEL", "RW", "RW",
"BEL", "PSC", "PSC",
"BEL", "FDF", "FDF",
"BEL", "VOLK", "VOLK",
"BEL", "PLP", "PLP",
# France (Fourth Republic)
"FRA", "PCF", "PCF",
"FRA", "SFIO", "SFIO",
"FRA", "MRP", "MRP",
"FRA", "RDA", "RDA",
"FRA", "UDSR", "UDSR",
"FRA", "RAD", "RAD",
"FRA", "RS", "RS",
"FRA", "RPF", "RPF",
"FRA", "AR", "AR",
"FRA", "ARS", "ARS",
"FRA", "RI", "RI",
"FRA", "CNIP", "CNIP",
"FRA", "PUS", "PUS",
"FRA", "PAYS", "PAYS",
"FRA", "AP", "AP",
"FRA", "PRL", "PRL",
"FRA", "POUJ", "POUJ",
# Weimar Germany
"DEU", "KPD", "KPD",
"DEU", "SDAP", "SDAP",
"DEU", "DDP", "DDP",
"DEU", "DZP", "DZP",
"DEU", "BVP", "BVP",
"DEU", "DVP", "DVP",
"DEU", "RDMW", "RDMW",
"DEU", "LVP", "LVP",
"DEU", "DNVP", "DNVP",
"DEU", "NAZI", "NAZI",
# Italy
"ITA", "PCI", "PCI",
"ITA", "PSIU", "PSIU",
"ITA", "PSI", "PSI",
"ITA", "PSDI", "PSDI",
"ITA", "PRI", "PRI",
"ITA", "DC", "DC",
"ITA", "PLI", "PLI",
"ITA", "MON", "MON",
"ITA", "MSI", "MSI",
# Luxembourg
"LUX", "COMM", "COMM",
"LUX", "SOCd", "SOCD",
"LUX", "CSOC", "CSOC",
"LUX", "GRPD", "GRPD",
# Israel
"ISR", "RAKA", "RAKA",
"ISR", "MAKI", "MAKI",
"ISR", "MAPM", "MAPM",
"ISR", "MADT", "MADT",
"ISR", "ADUT", "ADUT",
"ISR", "MAAR", "MAAR",
"ISR", "LAB", "LAB",
"ISR", "MAPI", "MAPI",
"ISR", "PAUG", "PAUG",
"ISR", "RAFI", "RAFI",
"ISR", "PROG", "PROG",
"ISR", "ILIB", "ILIB",
"ISR", "NRP", "NRP",
"ISR", "URF", "URF",
"ISR", "LIB", "LIB",
"ISR", "NATL", "NATL",
"ISR", "TORA", "TORA",
"ISR", "LIKD", "LIKD",
"ISR", "ZION", "ZION",
"ISR", "GHAL", "GHAL",
"ISR", "AGDT", "AGDT",
"ISR", "HRUT", "HRUT"
)
# Some parties in raw data that don't have exact matches - need special handling
# (e.g., parties that only exist in one period in PartyFacts but appear in both)
# We'll join using the period-based matching
# Expand periods to years for matching
morgan_expanded <- morgan_raw %>%
mutate(
year_start = as.integer(str_extract(period, "^\\d{4}")),
year_end = as.integer(str_extract(period, "\\d{4}$"))
)
# Join with abbreviation map
morgan_mapped <- morgan_expanded %>%
left_join(abbrev_map, by = c("country", "party_abbrev"))
# Check for unmatched abbreviations
unmatched_abbrev <- morgan_mapped %>%
filter(is.na(party_abbrev_pf)) %>%
distinct(country, party_abbrev)
if (nrow(unmatched_abbrev) > 0) {
cat("\nWarning: Unmatched abbreviations:\n")
print(unmatched_abbrev)
}
# Join with PartyFacts
morgan_joined <- morgan_mapped %>%
left_join(morgan_pf, by = c("country", "party_abbrev_pf")) %>%
# For parties with overlapping periods, use period overlap
mutate(
period_overlap = pmax(0,
pmin(year_end, year_last) - pmax(year_start, year_first) + 1)
) %>%
# Keep best match per party-period (max overlap)
group_by(country, party_abbrev, period) %>%
slice_max(period_overlap, n = 1, with_ties = FALSE) %>%
ungroup()
# Check for unmatched parties
unmatched <- morgan_joined %>%
filter(is.na(partyfacts_id)) %>%
distinct(country, party_abbrev, party_name, period)
if (nrow(unmatched) > 0) {
cat(sprintf("\n%d party-periods without PartyFacts match:\n", nrow(unmatched)))
print(unmatched)
}
# Dedup: when multiple abbreviations map to the same PF ID, keep only one
matched <- morgan_joined %>%
filter(!is.na(partyfacts_id)) %>%
group_by(country, partyfacts_id, period) %>%
slice(1) %>%
ungroup()
cat(sprintf("\nMatched %d of %d party-period observations (%.1f%%)\n",
nrow(matched), nrow(morgan_raw),
100 * nrow(matched) / nrow(morgan_raw)))
# Normalize position to [0,1] scale
# Original scale: 0-100
# Apply boundary adjustments like other expert data
eps <- 0.005
morgan_processed <- matched %>%
mutate(
# Normalize to [0,1]
lr_morgan = position / 100,
# Apply boundary adjustments
lr_morgan = case_when(
lr_morgan <= 0 ~ eps,
lr_morgan >= 1 ~ 1 - eps,
TRUE ~ lr_morgan
),
# Calculate standard error (sd / sqrt(n))
lr_morgan_se = (sd / 100) / sqrt(n_surveys),
# Set minimum SE for extreme parties (sd=0)
lr_morgan_se = pmax(lr_morgan_se, 0.01)
) %>%
select(
country,
partyfacts_id,
period,
year_start,
year_end,
party_abbrev,
party_name,
lr_morgan,
lr_morgan_se,
n_surveys
) %>%
arrange(country, year_start, lr_morgan)
# Summary statistics
cat("\nSummary of processed Morgan data:\n")
cat(sprintf(" Countries: %d\n", n_distinct(morgan_processed$country)))
cat(sprintf(" Parties: %d\n", n_distinct(morgan_processed$partyfacts_id)))
cat(sprintf(" Observations: %d\n", nrow(morgan_processed)))
# Distribution of positions
cat("\nPosition distribution:\n")
print(summary(morgan_processed$lr_morgan))
# Write output
write_csv(morgan_processed, "morgan_data.csv")
cat(sprintf("\nWrote morgan_data.csv with %d rows\n", nrow(morgan_processed)))
# Also provide a summary by country and period
summary_by_country <- morgan_processed %>%
group_by(country, period) %>%
summarise(
n_parties = n(),
mean_pos = mean(lr_morgan),
sd_pos = sd(lr_morgan),
.groups = "drop"
)
cat("\nSummary by country and period:\n")
print(summary_by_country, n = 50)
# ============================================================
# Generate lr_data-compatible output for pipeline integration
# ============================================================
cat("\n============================================================\n")
cat("Generating lr_data-compatible output (postwar only)\n")
cat("============================================================\n")
# Load text_data to get party-years with manifesto/PolDem coverage
if (!file.exists("text_data.csv")) {
cat("text_data.csv not present yet; skipping morgan_lr.csv generation on this pass.\n")
} else {
text_data <- read_csv("text_data.csv", show_col_types = FALSE)
# Convert Morgan ISO3 country codes to ISO2 (matching text_data format)
iso3_to_iso2 <- c(
"DNK" = "DK", "FIN" = "FI", "ISL" = "IS", "NOR" = "NO", "SWE" = "SE",
"NLD" = "NL", "BEL" = "BE", "DEU" = "DE", "FRA" = "FR", "ITA" = "IT",
"LUX" = "LU", "ISR" = "IL"
)
# Filter to postwar periods only (1945+)
morgan_postwar <- morgan_processed %>%
filter(year_end >= 1945) %>%
mutate(country_iso2 = iso3_to_iso2[country])
cat(sprintf("Postwar Morgan observations: %d party-periods\n", nrow(morgan_postwar)))
cat(sprintf("Countries: %s\n", paste(unique(morgan_postwar$country_iso2), collapse = ", ")))
# Get unique party-years from text_data
text_party_years <- text_data %>%
select(party, country, year) %>%
distinct()
cat(sprintf("Unique party-years in text_data: %d\n", nrow(text_party_years)))
# For each Morgan party-period, expand to all years where that party has text data
# within the Morgan period range (1945-1973 for postwar)
morgan_lr <- morgan_postwar %>%
# Join with text_data party-years
# Many-to-many is expected: one Morgan party-period maps to multiple years
inner_join(
text_party_years,
by = c("partyfacts_id" = "party", "country_iso2" = "country"),
relationship = "many-to-many"
) %>%
# Keep only years within the Morgan period
filter(year >= year_start & year <= year_end) %>%
# Format for lr_data.csv compatibility
transmute(
country = country_iso2,
party = partyfacts_id,
var = "lr_morgan",
year = year,
val = lr_morgan,
project = "Morgan",
# Morgan's continuous 0-100 scale is discretized to 10 points (matching CHES
# resolution) with the actual number of experts. The reconstructed sum
# round(mean × K × 10) is analogous to how CHES means are handled.
# See docs/k_scaling_validation.md Section 4.
n_scale = 10L,
val_int = as.integer(round(lr_morgan * 10)),
n_experts = as.integer(n_surveys)
) %>%
distinct() %>%
arrange(country, party, year)
cat(sprintf("\nGenerated %d lr_morgan observations\n", nrow(morgan_lr)))
cat(sprintf(" Unique parties: %d\n", n_distinct(morgan_lr$party)))
cat(sprintf(" Year range: %d-%d\n", min(morgan_lr$year), max(morgan_lr$year)))
# Summary by country
morgan_lr_summary <- morgan_lr %>%
group_by(country) %>%
summarise(
n_parties = n_distinct(party),
n_obs = n(),
year_min = min(year),
year_max = max(year),
.groups = "drop"
)
cat("\nMorgan L-R data by country:\n")
print(morgan_lr_summary, n = 20)
# Write morgan_lr.csv
write_csv(morgan_lr, "morgan_lr.csv")
cat(sprintf("\nWrote morgan_lr.csv with %d rows\n", nrow(morgan_lr)))
}
+162
View File
@@ -0,0 +1,162 @@
# ============================================================
# process_poldem.R - PolDem Media Data Processing
# ============================================================
# Processes PolDem (Political Deliberation in the Media) data
# for the two-dimensional party-position model
#
# Input: $PARTY2D_RAW_DATA_DIR/poldem/poldem-election_all.csv (sentence-level)
# Output: poldem_data.csv (party-year-var aggregates)
# ============================================================
library(tidyverse)
library(countrycode)
# Set working directory (works both in RStudio and command line)
if (interactive() && requireNamespace("rstudioapi", quietly = TRUE)) {
try(setwd(dirname(rstudioapi::getActiveDocumentContext()$path)), silent = TRUE)
}
cat("Processing PolDem media data...\n")
raw_data_dir <- Sys.getenv(
"PARTY2D_RAW_DATA_DIR",
unset = file.path("..", "..", "_local", "raw")
)
poldem_raw_path <- file.path(raw_data_dir, "poldem", "poldem-election_all.csv")
partyfacts_path <- file.path(raw_data_dir, "partyfacts", "partyfacts-external-parties.csv")
# ============================================================
# PartyFacts Linkage (via CMP party IDs)
# ============================================================
partyfacts_raw <- read_csv(partyfacts_path, show_col_types = FALSE)
manifesto_link <- partyfacts_raw %>%
filter(dataset_key == "manifesto") %>%
transmute(cmp = as.numeric(dataset_party_id), # Convert to numeric for join
country_pf = countrycode(country, origin = 'iso3c', destination = "iso2c"),
party = partyfacts_id,
party = ifelse(party == 622, 604, party))
# ============================================================
# Issue Category Mapping to 4 Dimensions
# ============================================================
# For positive direction: type_high is the active trait
# For negative direction: we flip (same data, just contributes to the opposite trait)
poldem_mapping <- tribble(
~issue_cat, ~dimension, ~type_high, ~type_low,
# Economic dimension
"ecolib", "economic", "pro_market", "pro_welfare", # Economic liberalization
"welfare", "economic", "pro_welfare", "pro_market", # Welfare state
# Final exclusion: the PolDem economic-reform category is intentionally
# omitted because item-response diagnostics showed that it did not load
# substantively onto the economic latent trait.
# Cultural dimension
"immig", "cultural", "cosmopolitan", "traditional", # Immigration (pro = cosmopolitan)
"cultlib", "cultural", "cosmopolitan", "traditional", # Cultural liberalism
"nationalism", "cultural", "traditional", "cosmopolitan", # Nationalism (pro = traditional)
"europe", "cultural", "cosmopolitan", "traditional", # EU integration (pro = cosmopolitan)
"euro", "cultural", "cosmopolitan", "traditional", # Euro currency (pro = cosmopolitan)
"defense", "cultural", "traditional", "cosmopolitan", # Defense (pro = traditional)
"security", "cultural", "traditional", "cosmopolitan" # Security/law-order (pro = traditional)
)
cat(sprintf(" Using %d issue categories\n", nrow(poldem_mapping)))
# ============================================================
# Load and Clean PolDem Data
# ============================================================
poldem_raw <- read_csv(poldem_raw_path, show_col_types = FALSE)
cat(sprintf(" Raw PolDem data: %d rows\n", nrow(poldem_raw)))
poldem <- poldem_raw %>%
# Fix country codes
mutate(country = case_when(
iso2code == "AU" ~ "AT", # Austria (PolDem uses AU instead of AT)
iso2code == "UK" ~ "GB", # United Kingdom
TRUE ~ iso2code
)) %>%
# Extract year from article date (format: YYYY-MM-DD)
mutate(year = suppressWarnings(as.numeric(substr(date_art, 1, 4)))) %>%
# Filter to valid issue categories only
filter(issue_cat %in% poldem_mapping$issue_cat) %>%
# Convert direction to numeric and filter out neutral/NA
mutate(direction = as.numeric(direction)) %>%
filter(!is.na(direction) & direction != 0) %>%
# Remove rows with invalid years
filter(!is.na(year))
cat(sprintf(" After filtering: %d rows (valid issues, non-neutral)\n", nrow(poldem)))
# ============================================================
# Link to PartyFacts via CMP codes
# ============================================================
poldem <- poldem %>%
mutate(cmp = as.numeric(cmp)) %>%
left_join(manifesto_link, by = "cmp") %>%
filter(!is.na(party))
# Report linkage
n_linked <- nrow(poldem)
n_unlinked <- nrow(poldem_raw %>%
filter(issue_cat %in% poldem_mapping$issue_cat) %>%
mutate(direction = as.numeric(direction)) %>%
filter(!is.na(direction) & direction != 0)) - n_linked
cat(sprintf(" Linked to PartyFacts: %d rows\n", n_linked))
if (n_unlinked > 0) {
cat(sprintf(" Warning: %d rows could not be linked (missing CMP mapping)\n", n_unlinked))
}
# ============================================================
# Aggregate to Party-Year-Issue Level
# Using round(sum()) for weak direction values (0.5, -0.5)
# ============================================================
poldem_agg <- poldem %>%
left_join(poldem_mapping, by = "issue_cat") %>%
group_by(party, country, year, issue_cat, type_high, type_low) %>%
summarise(
# Sum positive directions (0.5 and 1), then round
positive = round(sum(direction[direction > 0])),
# Sum absolute directions for sample (all non-neutral), then round
sample = round(sum(abs(direction))),
n_obs = n(),
.groups = "drop"
) %>%
# Minimum 3 observations per group
filter(n_obs >= 3) %>%
select(-n_obs)
cat(sprintf(" After aggregation: %d party-year-issue observations\n", nrow(poldem_agg)))
# ============================================================
# Format Output (matching manifesto structure)
# ============================================================
poldem_data <- poldem_agg %>%
mutate(
var = paste0(issue_cat, "_poldem"),
project = "PolDem"
) %>%
select(party, country, year, var, positive, sample, type_high, type_low, project)
# ============================================================
# Write Output
# ============================================================
write_csv(poldem_data, "poldem_data.csv")
cat(sprintf("\nOutput: poldem_data.csv\n"))
cat(sprintf(" Total rows: %d\n", nrow(poldem_data)))
cat(sprintf(" Unique parties: %d\n", n_distinct(poldem_data$party)))
cat(sprintf(" Countries: %s\n", paste(sort(unique(poldem_data$country)), collapse = ", ")))
cat(sprintf(" Year range: %d-%d\n", min(poldem_data$year, na.rm = TRUE), max(poldem_data$year, na.rm = TRUE)))
cat("\n Rows by issue category:\n")
poldem_data %>%
group_by(var) %>%
summarise(n = n(), .groups = "drop") %>%
arrange(desc(n)) %>%
print()
+86
View File
@@ -0,0 +1,86 @@
# Data setup
This directory exists because the public repository cannot redistribute the original raw/source files. It downloads script-accessible sources, checks user-provided source files, rebuilds model-ready inputs, and compares them with the committed files in `../data/`.
The main estimation workflow does not run these scripts. Once the five model-ready CSVs exist in `data/`, fitting and post-estimation are Julia/Stan-only.
The setup workflow never overwrites committed files in `data/`.
Put raw source files in `_local/raw/` or set `PARTY2D_RAW_DATA_DIR` to another local directory. See `source_manifest.csv` and the source notes below for source-specific access requirements.
## What downloads automatically?
`data-setup/R/01_download_sources.R` downloads source files that are script-accessible under the providers' terms and reports sources that require credentials or manual local files.
| Source | Automatic? | Requirement |
| --- | --- | --- |
| PolDem | Yes | none |
| PartyFacts crosswalk | Yes | none |
| CHES family files | Yes | none where provider links are live |
| POPPA | Yes | none; downloaded from Harvard Dataverse |
| Global Party Survey 2019 | Yes | none; downloaded from Harvard Dataverse |
| V-Party | Yes, with provider form details | set `PARTY2D_VDEM_EMAIL`; optionally set `PARTY2D_VDEM_GENDER` |
| Manifesto Project | Yes, with credentials | set your own `MANIFESTO_API_KEY` or `PARTY2D_MANIFESTO_API_KEY` |
| Morgan historical expert data | No | place `morgan_positions_raw.csv` locally; available on request |
Do not commit downloaded or user-provided source files.
Recommended local layout:
```text
_local/raw/
manifesto/MPDataset_MPDS2025a.csv
poldem/poldem-election_all.csv
ches/...
vparty/...
poppa/...
gps/...
morgan/...
partyfacts/partyfacts-external-parties.csv
```
Local output layout:
```text
_local/build/ # intermediate processing files
_local/generated-inputs/ # regenerated final model-input CSVs
_local/reports/ # comparison reports
```
## Commands
Run the full source-data setup workflow with:
```bash
bash data-setup/run_data_setup.sh
```
The command downloads script-accessible sources, checks required local files,
rebuilds model-ready inputs under `_local/generated-inputs/`, and writes a
comparison report under `_local/reports/`. It never replaces committed files in
`data/`.
V-Party:
```bash
export PARTY2D_VDEM_EMAIL='you@example.org'
export PARTY2D_VDEM_GENDER='' # blank means prefer not to say
```
Manifesto Project:
```bash
export MANIFESTO_API_KEY='...'
# or
export PARTY2D_MANIFESTO_API_KEY='...'
```
Morgan is not a public provider download. The local OCR/transcription file can be provided on request and should be placed at:
```text
_local/raw/morgan/morgan_positions_raw.csv
```
The comparison writes `_local/reports/input_comparison.md`. Replacing committed inputs, if ever needed, is a separate manual decision and is not done by these scripts.
Known behavior: the current public/rebuilt sources run through the workflow successfully, but regenerated `text_data.csv`, `expert.csv`, and `lr_data.csv` are not byte-identical to the committed model-ready inputs because of source-version and linkage differences. The comparison report records those differences explicitly.
+58
View File
@@ -0,0 +1,58 @@
#!/usr/bin/env bash
set -euo pipefail
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P)"
raw_data_dir="${PARTY2D_RAW_DATA_DIR:-$repo_root/_local/raw}"
report_dir="${PARTY2D_REPORT_DIR:-$repo_root/_local/reports}"
mkdir -p "$report_dir"
missing_report="$report_dir/raw_data_preflight_missing.txt"
: > "$missing_report"
required_files=(
"manifesto/MPDataset_MPDS2025a.csv"
"poldem/poldem-election_all.csv"
"partyfacts/partyfacts-external-parties.csv"
"ches/1999-2019_CHES_dataset_means(v3).csv"
"ches/CHES_2024_final_v2.csv"
"ches/CHES_2024_expert_level.csv"
"ches/CHES_CA2023.csv"
"ches/CHES_CA2023_expert_level.csv"
"ches/ches_la_2020_aggregate_level_v01.csv"
"ches/CHES_LA2020_expert_level.csv"
"ches/CHES_ISRAEL_means_2021_2022.csv"
"ches/CHES_IL_expert_level.csv"
"vparty/V-Dem-CPD-Party-V2.rds"
"poppa/poppa_integrated_v2.rds"
"gps/Global Party Survey by Party SPSS V2_1_Apr_2020-2.tab"
"morgan/morgan_positions_raw.csv"
)
echo "Raw data directory: $raw_data_dir"
echo
echo "Required raw inputs for regeneration:"
missing=0
for rel in "${required_files[@]}"; do
path="$raw_data_dir/$rel"
if [ -s "$path" ]; then
bytes="$(wc -c < "$path")"
read -r sha _ < <(sha256sum "$path")
echo " OK $rel ($bytes bytes, sha256=$sha)"
else
echo " MISSING $rel"
printf '%s\n' "$rel" >> "$missing_report"
missing=1
fi
done
if [ "$missing" -ne 0 ]; then
echo
echo "At least one required raw input is missing." >&2
echo "Missing-file report: $missing_report" >&2
echo "See data-setup/README.md and docs/RAW_DATA_SOURCES.md for instructions." >&2
exit 1
fi
echo
echo "Required raw data preflight passed."
rm -f "$missing_report"
+34
View File
@@ -0,0 +1,34 @@
#!/usr/bin/env bash
set -euo pipefail
usage() {
cat >&2 <<'EOF'
Usage: bash data-setup/run_data_setup.sh
EOF
}
if [ "$#" -ne 0 ]; then
usage
exit 1
fi
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P)"
cd "$repo_root"
export PARTY2D_RAW_DATA_DIR="${PARTY2D_RAW_DATA_DIR:-$repo_root/_local/raw}"
export PARTY2D_BUILD_DIR="${PARTY2D_BUILD_DIR:-$repo_root/_local/build}"
export PARTY2D_GENERATED_INPUT_DIR="${PARTY2D_GENERATED_INPUT_DIR:-$repo_root/_local/generated-inputs}"
export PARTY2D_REPORT_DIR="${PARTY2D_REPORT_DIR:-$repo_root/_local/reports}"
export R_LIBS_USER="${R_LIBS_USER:-$repo_root/_local/R/library}"
mkdir -p "$R_LIBS_USER"
command -v bash >/dev/null
command -v Rscript >/dev/null
bash -n data-setup/check_raw_data.sh
Rscript data-setup/R/00_install_dependencies.R
Rscript data-setup/R/01_download_sources.R || true
bash data-setup/check_raw_data.sh
rm -rf "$PARTY2D_BUILD_DIR" "$PARTY2D_GENERATED_INPUT_DIR"
mkdir -p "$PARTY2D_BUILD_DIR" "$PARTY2D_GENERATED_INPUT_DIR"
Rscript data-setup/R/02_build_model_inputs.R
Rscript data-setup/R/03_compare_generated_inputs.R
+17
View File
@@ -0,0 +1,17 @@
source,scope,local_path,access,automatic_download,notes
Manifesto Project,manifesto text/coding,manifesto/MPDataset_MPDS2025a.csv,API key/login required,yes with MANIFESTO_API_KEY,Use the MPDS 2025a CSV export; raw file is not redistributed.
PolDem Election Campaigns,media campaign issue statements,poldem/poldem-election_all.csv,public download,yes,CSV URL https://poldem.eui.eu/downloads/cosa/poldem-election_all.csv; observed sha256 2cd8c9108b1b0b9c1b6594bb21acee709c70259cd02f450bc69fc09b505fc9fb.
CHES 1999-2019,expert party placements,ches/1999-2019_CHES_dataset_means(v3).csv,public archived download,yes,Downloaded from archived CHES URL at chesdata.eu.
CHES 2024,expert party placements,ches/CHES_2024_final_v2.csv,CHES terms,no,Requires matching expert-level file for expert counts.
CHES 2024 expert level,expert counts,ches/CHES_2024_expert_level.csv,CHES terms,no,Required for expert counts.
CHES Canada 2023 aggregate,expert party placements,ches/CHES_CA2023.csv,CHES terms,no,Required for Canada extension.
CHES Canada 2023,expert party placements,ches/CHES_CA2023_expert_level.csv,CHES terms,no,Used by expert-source processing where available.
CHES Latin America aggregate,expert party placements,ches/ches_la_2020_aggregate_level_v01.csv,CHES terms,no,Required for Latin America extension.
CHES Latin America 2020,expert party placements,ches/CHES_LA2020_expert_level.csv,CHES terms,no,Used by expert-source processing where available.
CHES Israel aggregate,expert party placements,ches/CHES_ISRAEL_means_2021_2022.csv,CHES terms,no,Required for Israel extension.
CHES Israel,expert party placements,ches/CHES_IL_expert_level.csv,CHES terms,no,Used by expert-source processing where available.
V-Party,expert-coded party variables,vparty/V-Dem-CPD-Party-V2.rds,V-Dem form terms,yes with PARTY2D_VDEM_EMAIL,Downloader submits provider form and extracts R data from ZIP.
POPPA,expert party placements,poppa/poppa_integrated_v2.rds,public Dataverse,yes,Downloaded from Harvard Dataverse DOI 10.7910/DVN/RMQREQ.
Global Party Survey 2019,expert party placements,gps/Global Party Survey by Party SPSS V2_1_Apr_2020-2.tab,public Dataverse,yes,Downloaded from Harvard Dataverse DOI 10.7910/DVN/WMGTNS.
Morgan historical expert data,historical left-right placements,morgan/morgan_positions_raw.csv,derived local transcription/no redistribution,no public URL,Local OCR/transcription source used for historical anchoring.
PartyFacts crosswalk,party ID harmonization,partyfacts/partyfacts-external-parties.csv,public download,yes,Support crosswalk required by source-processing scripts.
1 source scope local_path access automatic_download notes
2 Manifesto Project manifesto text/coding manifesto/MPDataset_MPDS2025a.csv API key/login required yes with MANIFESTO_API_KEY Use the MPDS 2025a CSV export; raw file is not redistributed.
3 PolDem Election Campaigns media campaign issue statements poldem/poldem-election_all.csv public download yes CSV URL https://poldem.eui.eu/downloads/cosa/poldem-election_all.csv; observed sha256 2cd8c9108b1b0b9c1b6594bb21acee709c70259cd02f450bc69fc09b505fc9fb.
4 CHES 1999-2019 expert party placements ches/1999-2019_CHES_dataset_means(v3).csv public archived download yes Downloaded from archived CHES URL at chesdata.eu.
5 CHES 2024 expert party placements ches/CHES_2024_final_v2.csv CHES terms no Requires matching expert-level file for expert counts.
6 CHES 2024 expert level expert counts ches/CHES_2024_expert_level.csv CHES terms no Required for expert counts.
7 CHES Canada 2023 aggregate expert party placements ches/CHES_CA2023.csv CHES terms no Required for Canada extension.
8 CHES Canada 2023 expert party placements ches/CHES_CA2023_expert_level.csv CHES terms no Used by expert-source processing where available.
9 CHES Latin America aggregate expert party placements ches/ches_la_2020_aggregate_level_v01.csv CHES terms no Required for Latin America extension.
10 CHES Latin America 2020 expert party placements ches/CHES_LA2020_expert_level.csv CHES terms no Used by expert-source processing where available.
11 CHES Israel aggregate expert party placements ches/CHES_ISRAEL_means_2021_2022.csv CHES terms no Required for Israel extension.
12 CHES Israel expert party placements ches/CHES_IL_expert_level.csv CHES terms no Used by expert-source processing where available.
13 V-Party expert-coded party variables vparty/V-Dem-CPD-Party-V2.rds V-Dem form terms yes with PARTY2D_VDEM_EMAIL Downloader submits provider form and extracts R data from ZIP.
14 POPPA expert party placements poppa/poppa_integrated_v2.rds public Dataverse yes Downloaded from Harvard Dataverse DOI 10.7910/DVN/RMQREQ.
15 Global Party Survey 2019 expert party placements gps/Global Party Survey by Party SPSS V2_1_Apr_2020-2.tab public Dataverse yes Downloaded from Harvard Dataverse DOI 10.7910/DVN/WMGTNS.
16 Morgan historical expert data historical left-right placements morgan/morgan_positions_raw.csv derived local transcription/no redistribution no public URL Local OCR/transcription source used for historical anchoring.
17 PartyFacts crosswalk party ID harmonization partyfacts/partyfacts-external-parties.csv public download yes Support crosswalk required by source-processing scripts.
+13
View File
@@ -0,0 +1,13 @@
# Data directory
This directory contains only the processed, model-ready inputs used by the Julia/Stan estimation pipeline:
- `text_data.csv`
- `expert.csv`
- `lr_data.csv`
- `union_mapping.csv`
- `party_families.csv`
Original raw source files and intermediate build products are not stored here. To regenerate the processed inputs, place raw files in a local directory and set `PARTY2D_RAW_DATA_DIR`; see `../data-setup/README.md`.
Generated outputs and temporary staging files are ignored by git.
+25079
View File
File diff suppressed because it is too large Load Diff
+2208
View File
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+3
View File
@@ -0,0 +1,3 @@
6b9aeda20489000983e459e4eb19d2788ed035eaab5deb9a5e010338670b2ff1 party_2d_election_year_panel_v0.zip
a7ba39d1b35f6b21d98a92f779e64679646f0d245ea560a3f2ee551c97066280 party_2d_annual_model_output_v0.csv.xz
b99f7ef0e8a4c821183a2a0f957752303479e78e5afb685f173fd17f60039b3e party_2d_diagnostics_report_v0.pdf
Binary file not shown.
+38203
View File
File diff suppressed because it is too large Load Diff
+107
View File
@@ -0,0 +1,107 @@
manifesto_pf_id,manifesto_name,expert_pf_id,expert_name,country,status
3889,PJ,6648,PF-PJ,AR,implemented
6161,FAP,1365,PS,AR,implemented
6161,FAP,6160,FR,AR,implemented
6161,FAP,6554,FPCyS,AR,implemented
486,LPA,1998,LP,AU,implemented
1743,NPA,338,NAT,AU,implemented
1760,HDZ BiH,3904,HDZ-HK~HNZ,BA,implemented
36,N-VA,756,CD+NVA,BE,implemented
604,CD/V,622,CD&V,BE,implemented
604,CD/V,756,CD+NVA,BE,implemented
1680,sp.a,1586,sp.a-SPIRIT,BE,implemented
374,NDSV,5848,KSII,BG,implemented
482,SDS,3908,G-VMRO; VMRO-BND,BG,implemented
1765,ONS,3908,G-VMRO; VMRO-BND,BG,implemented
5649,Patriotic Front - NFSB and VMRO,2057,NFSB,BG,implemented
5649,Patriotic Front - NFSB and VMRO,3908,G-VMRO; VMRO-BND,BG,implemented
360,FDP/PLR,1231,FDP/PLR,CH,implemented
6061,Alliance,928,RN,CL,implemented
6061,Alliance,1599,UDI,CL,implemented
1707,STAN,751,SNK-ED,CZ,implemented
2138,LB,1041,KSC,CZ,implemented
6202,KDU-ČSL-US-DEU,104,US-DEU,CZ,implemented
211,CDU/CSU,1375,CDU,DE,implemented
211,CDU/CSU,1731,CSU,DE,implemented
1816,B90/Grüne,10,Die Grünen,DE,implemented
3925,RED-ID,797,ID,EC,implemented
685,RP,491,ERP,EE,implemented
779,I/ERSP,908,RKI,EE,implemented
779,I/ERSP,1299,ERSP,EE,implemented
139,CiU,4795,CDC,ES,implemented
8271,Compromís–Podemos–EUPV,5623,CC,ES,implemented
213,MoDem,496,MoDem,FR,implemented
1108,EELV,5650,EELV,FR,implemented
1595,UMP,4628,Les Républicains,FR,implemented
1595,UMP,8168,LR,FR,implemented
1468,PASOK,7909,KINAL,GR,implemented
7347,EL,378,OP,GR,implemented
1475,SDP,8842,SDP-HSLS,HR,implemented
2522,Kukuriku,78,DC,HR,implemented
3648,ZL,8036,HKDU,HR,implemented
3918,DA-IDS-RDS,513,IDS,HR,implemented
242,PBP,8241,PBPS,IE,implemented
201,UdC,1758,UC,IT,implemented
1212,SEL,7031,SEL,IT,implemented
1737,Olive Tree,878,DS,IT,implemented
6241,House of Freedom,813,AN,IT,implemented
6241,House of Freedom,1519,CeD,IT,implemented
1967,SLFP,4020,CP / VLSSP,LK,implemented
1967,SLFP,5414,LSS,LK,implemented
1967,SLFP,6691,CP,LK,implemented
197,BSDK,168,LRS,LT,implemented
197,BSDK,1747,LMP-NDP,LT,implemented
377,LTS,1410,LLaS,LT,implemented
5779,SK,1407,LZP,LT,implemented
186,LSAP/POSL,898,SDP,LU,implemented
708,LNNK-LZP,1296,LZP,LV,implemented
1704,TB-LNNK,671,TB,LV,implemented
1704,TB-LNNK,1789,LNNK,LV,implemented
1704,TB-LNNK,7619,NATBLNNK,LV,implemented
7622,ACUM,7904,PAS,MD,implemented
3254,DSCG,3253,HGI,ME,implemented
1537,GL,1533,Groen,NL,implemented
716,Alliance,1119,NLP,NZ,implemented
4219,C90,5130,P2000,PE,implemented
1458,WAK,70,ZChN,PL,implemented
8268,UW,1566,D|W|U,PL,implemented
192,CDR,645,PAC,RO,implemented
1347,PSD-PUR,1443,PU|PC,RO,implemented
5941,USL,120,PSD,RO,implemented
5941,USL,481,PNL,RO,implemented
5941,USL,1541,UNPR,RO,implemented
6153,PSD-PC,1443,PU|PC,RO,implemented
8626,LDP/LSV/SDS,4769,LSV,RS,implemented
205,SV,200,SDSS,SK,implemented
226,SDK,200,SDSS,SK,implemented
1617,SDKÚ-DS,983,DS,SK,implemented
6629,DÚS,707,DUS,SK,implemented
1658,FA,3671,NE,UY,implemented
301,"SYRIZA, SYN; SYRIZA, Syriza, SYN/SYRIZA",1682,DIKKI,GR,implemented
676,"KDU, KDU-ČSL, KDU-CSL, KDU/CSL, KDUCSL, KDU–CSL, KDU–Č, KDU- ČSL, CSL",824,KDS,CZ,implemented
701,"ZZS, LZS",1702,LZS,LV,implemented
852,"V, Unity, UNITY, VIENOTIBA, JV, PS",1531,JL,LV,implemented
2190,"DSS, DSS/NS",2346,NS,RS,implemented
2530,"FpV, FPV, FPV-PJ, AFplV, FplV",623,PJ,AR,implemented
3906,NA,356,PT,BR,implemented
3906,NA,723,PSB,BR,implemented
3906,NA,1009,PDT,BR,implemented
3906,NA,4405,"PR, PR (2), PR / PL, PR/PL, PR PL, PL/PR",BR,implemented
3906,NA,458,PTB,BR,implemented
3906,NA,1823,PL,BR,implemented
4550,"C, Concertacion, CPD",6,PS,CL,implemented
4550,"C, Concertacion, CPD",54,PPD,CL,implemented
4550,"C, Concertacion, CPD",390,PDC,CL,implemented
4550,"C, Concertacion, CPD",437,PRSD,CL,implemented
8999,"ZMS, Aleksandar V..., PS-TN, Serbia is Wi..., ally",3177,SNS,RS,implemented
1117,PO,4630,.N,PL,implemented
4550,Concertacion,162,PC,CL,implemented
4550,Concertacion,209,PH,CL,implemented
5668,EH Bildu,1671,Amaiur,ES,implemented
6241,CdL,1767,CCD,IT,implemented
3979,Salvemos a México,1474,PRI,MX,implemented
3979,Salvemos a México,446,PVEM,MX,implemented
7912,Joint List,421,Hadash,IL,implemented
7912,Joint List,1663,Balad,IL,implemented
365,PdL,1626,FI,IT,implemented
365,PdL,813,AN,IT,implemented
1 manifesto_pf_id manifesto_name expert_pf_id expert_name country status
2 3889 PJ 6648 PF-PJ AR implemented
3 6161 FAP 1365 PS AR implemented
4 6161 FAP 6160 FR AR implemented
5 6161 FAP 6554 FPCyS AR implemented
6 486 LPA 1998 LP AU implemented
7 1743 NPA 338 NAT AU implemented
8 1760 HDZ BiH 3904 HDZ-HK~HNZ BA implemented
9 36 N-VA 756 CD+NVA BE implemented
10 604 CD/V 622 CD&V BE implemented
11 604 CD/V 756 CD+NVA BE implemented
12 1680 sp.a 1586 sp.a-SPIRIT BE implemented
13 374 NDSV 5848 KSII BG implemented
14 482 SDS 3908 G-VMRO; VMRO-BND BG implemented
15 1765 ONS 3908 G-VMRO; VMRO-BND BG implemented
16 5649 Patriotic Front - NFSB and VMRO 2057 NFSB BG implemented
17 5649 Patriotic Front - NFSB and VMRO 3908 G-VMRO; VMRO-BND BG implemented
18 360 FDP/PLR 1231 FDP/PLR CH implemented
19 6061 Alliance 928 RN CL implemented
20 6061 Alliance 1599 UDI CL implemented
21 1707 STAN 751 SNK-ED CZ implemented
22 2138 LB 1041 KSC CZ implemented
23 6202 KDU-ČSL-US-DEU 104 US-DEU CZ implemented
24 211 CDU/CSU 1375 CDU DE implemented
25 211 CDU/CSU 1731 CSU DE implemented
26 1816 B90/Grüne 10 Die Grünen DE implemented
27 3925 RED-ID 797 ID EC implemented
28 685 RP 491 ERP EE implemented
29 779 I/ERSP 908 RKI EE implemented
30 779 I/ERSP 1299 ERSP EE implemented
31 139 CiU 4795 CDC ES implemented
32 8271 Compromís–Podemos–EUPV 5623 CC ES implemented
33 213 MoDem 496 MoDem FR implemented
34 1108 EELV 5650 EELV FR implemented
35 1595 UMP 4628 Les Républicains FR implemented
36 1595 UMP 8168 LR FR implemented
37 1468 PASOK 7909 KINAL GR implemented
38 7347 EL 378 OP GR implemented
39 1475 SDP 8842 SDP-HSLS HR implemented
40 2522 Kukuriku 78 DC HR implemented
41 3648 ZL 8036 HKDU HR implemented
42 3918 DA-IDS-RDS 513 IDS HR implemented
43 242 PBP 8241 PBPS IE implemented
44 201 UdC 1758 UC IT implemented
45 1212 SEL 7031 SEL IT implemented
46 1737 Olive Tree 878 DS IT implemented
47 6241 House of Freedom 813 AN IT implemented
48 6241 House of Freedom 1519 CeD IT implemented
49 1967 SLFP 4020 CP / VLSSP LK implemented
50 1967 SLFP 5414 LSS LK implemented
51 1967 SLFP 6691 CP LK implemented
52 197 BSDK 168 LRS LT implemented
53 197 BSDK 1747 LMP-NDP LT implemented
54 377 LTS 1410 LLaS LT implemented
55 5779 SK 1407 LZP LT implemented
56 186 LSAP/POSL 898 SDP LU implemented
57 708 LNNK-LZP 1296 LZP LV implemented
58 1704 TB-LNNK 671 TB LV implemented
59 1704 TB-LNNK 1789 LNNK LV implemented
60 1704 TB-LNNK 7619 NATBLNNK LV implemented
61 7622 ACUM 7904 PAS MD implemented
62 3254 DSCG 3253 HGI ME implemented
63 1537 GL 1533 Groen NL implemented
64 716 Alliance 1119 NLP NZ implemented
65 4219 C90 5130 P2000 PE implemented
66 1458 WAK 70 ZChN PL implemented
67 8268 UW 1566 D|W|U PL implemented
68 192 CDR 645 PAC RO implemented
69 1347 PSD-PUR 1443 PU|PC RO implemented
70 5941 USL 120 PSD RO implemented
71 5941 USL 481 PNL RO implemented
72 5941 USL 1541 UNPR RO implemented
73 6153 PSD-PC 1443 PU|PC RO implemented
74 8626 LDP/LSV/SDS 4769 LSV RS implemented
75 205 SV 200 SDSS SK implemented
76 226 SDK 200 SDSS SK implemented
77 1617 SDKÚ-DS 983 DS SK implemented
78 6629 DÚS 707 DUS SK implemented
79 1658 FA 3671 NE UY implemented
80 301 SYRIZA, SYN; SYRIZA, Syriza, SYN/SYRIZA 1682 DIKKI GR implemented
81 676 KDU, KDU-ČSL, KDU-CSL, KDU/CSL, KDUCSL, KDU–CSL, KDU–Č, KDU- ČSL, CSL 824 KDS CZ implemented
82 701 ZZS, LZS 1702 LZS LV implemented
83 852 V, Unity, UNITY, VIENOTIBA, JV, PS 1531 JL LV implemented
84 2190 DSS, DSS/NS 2346 NS RS implemented
85 2530 FpV, FPV, FPV-PJ, AFplV, FplV 623 PJ AR implemented
86 3906 NA 356 PT BR implemented
87 3906 NA 723 PSB BR implemented
88 3906 NA 1009 PDT BR implemented
89 3906 NA 4405 PR, PR (2), PR / PL, PR/PL, PR PL, PL/PR BR implemented
90 3906 NA 458 PTB BR implemented
91 3906 NA 1823 PL BR implemented
92 4550 C, Concertacion, CPD 6 PS CL implemented
93 4550 C, Concertacion, CPD 54 PPD CL implemented
94 4550 C, Concertacion, CPD 390 PDC CL implemented
95 4550 C, Concertacion, CPD 437 PRSD CL implemented
96 8999 ZMS, Aleksandar V..., PS-TN, Serbia is Wi..., ally 3177 SNS RS implemented
97 1117 PO 4630 .N PL implemented
98 4550 Concertacion 162 PC CL implemented
99 4550 Concertacion 209 PH CL implemented
100 5668 EH Bildu 1671 Amaiur ES implemented
101 6241 CdL 1767 CCD IT implemented
102 3979 Salvemos a México 1474 PRI MX implemented
103 3979 Salvemos a México 446 PVEM MX implemented
104 7912 Joint List 421 Hadash IL implemented
105 7912 Joint List 1663 Balad IL implemented
106 365 PdL 1626 FI IT implemented
107 365 PdL 813 AN IT implemented
+17
View File
@@ -0,0 +1,17 @@
# Diagnostics
This folder contains the repository diagnostics report for the Scientific Data Data Descriptor. It is generated from the model-ready inputs and the completed model/post-estimation outputs, so it can only be rerun after the estimation workflow has produced party-position, convergence, and validation artifacts.
Regenerate from the repository root with:
```bash
Rscript diagnostics/generate_diagnostics.R
```
If model outputs are stored outside the repository root, point the script to them:
```bash
PARTY2D_OUTPUTS_DIR=/path/to/outputs Rscript diagnostics/generate_diagnostics.R
```
Generated files are written to `diagnostics/generated/`. The PDF report is also copied to `data/releases/` for the release bundle. PDF rendering uses R Markdown/Pandoc and requires a LaTeX engine such as `pdflatex`.
+609
View File
@@ -0,0 +1,609 @@
#!/usr/bin/env Rscript
suppressPackageStartupMessages(library(tidyverse))
find_repo_root <- function() {
args <- commandArgs(trailingOnly = FALSE)
file_arg <- "--file="
script_arg <- args[startsWith(args, file_arg)][1]
if (!is.na(script_arg)) {
return(normalizePath(file.path(dirname(sub(file_arg, "", script_arg)), "..")))
}
if (file.exists("data/text_data.csv")) return(normalizePath(getwd()))
stop("Cannot find repository root. Run from the party2d repository root.")
}
repo_root <- find_repo_root()
setwd(repo_root)
release_version <- Sys.getenv("PARTY2D_RELEASE_VERSION", "v0")
outputs_dir <- Sys.getenv("PARTY2D_OUTPUTS_DIR", "outputs")
if (!grepl("^/", outputs_dir)) outputs_dir <- file.path(repo_root, outputs_dir)
if (!dir.exists(outputs_dir)) {
stop("Model output directory not found: ", outputs_dir, ". Run estimation/validation first or set PARTY2D_OUTPUTS_DIR.")
}
supplementary_inputs_dir <- Sys.getenv("PARTY2D_SUPPLEMENTARY_INPUTS_DIR", file.path(dirname(repo_root), "archive", "supplementary_inputs"))
if (!grepl("^/", supplementary_inputs_dir)) supplementary_inputs_dir <- file.path(repo_root, supplementary_inputs_dir)
generated_dir <- file.path(repo_root, "diagnostics", "generated")
release_dir <- file.path(repo_root, "data", "releases")
dir.create(generated_dir, recursive = TRUE, showWarnings = FALSE)
dir.create(release_dir, recursive = TRUE, showWarnings = FALSE)
required_inputs <- c(
"data/text_data.csv",
"data/expert.csv",
"data/lr_data.csv",
"data/union_mapping.csv",
"data/party_families.csv"
)
missing_inputs <- required_inputs[!file.exists(required_inputs)]
if (length(missing_inputs) > 0) {
stop("Missing required model input files: ", paste(missing_inputs, collapse = ", "))
}
latest_file <- function(path, pattern) {
if (!dir.exists(path)) return(NA_character_)
files <- list.files(path, pattern = pattern, full.names = TRUE)
if (length(files) == 0) return(NA_character_)
sort(files)[length(files)]
}
read_if_exists <- function(path) {
if (is.na(path) || !file.exists(path)) return(tibble())
readr::read_csv(path, show_col_types = FALSE)
}
supplementary_file <- function(...) {
path <- file.path(supplementary_inputs_dir, ...)
if (file.exists(path)) path else NA_character_
}
public_dimension <- function(x) {
dplyr::recode(as.character(x),
economic_lr = "economic left-right",
galtan = "cultural cosmopolitan--traditionalist",
Economic = "economic left-right",
Cultural = "cultural cosmopolitan--traditionalist",
`Economic Left-Right` = "economic left-right",
.default = as.character(x)
)
}
infer_dimension <- function(type_low, type_high) {
dplyr::case_when(
type_low %in% c("pro_market", "pro_welfare", "left", "right") |
type_high %in% c("pro_market", "pro_welfare", "left", "right") ~ "economic left-right",
type_low %in% c("cosmopolitan", "traditional") |
type_high %in% c("cosmopolitan", "traditional") ~ "cultural cosmopolitan--traditionalist",
TRUE ~ "general left-right"
)
}
is_reversed_for_reporting <- function(type_high) {
type_high %in% c("pro_welfare", "left", "cosmopolitan")
}
fmt_num <- function(x, digits = 3) {
ifelse(is.na(x), "NA", formatC(x, digits = digits, format = "f"))
}
display_path <- function(path) {
if (is.na(path) || !nzchar(path)) return("not available")
normalized <- normalizePath(path, mustWork = FALSE)
root_prefix <- paste0(normalizePath(repo_root, mustWork = FALSE), .Platform$file.sep)
if (startsWith(normalized, root_prefix)) return(sub(root_prefix, "", normalized, fixed = TRUE))
basename(path)
}
md_table <- function(df, n = Inf) {
if (nrow(df) == 0) return("_No rows available._\n")
df <- head(df, n)
df <- mutate(df, across(everything(), as.character))
header <- paste0("| ", paste(names(df), collapse = " | "), " |")
sep <- paste0("| ", paste(rep("---", ncol(df)), collapse = " | "), " |")
rows <- apply(df, 1, function(x) paste0("| ", paste(x, collapse = " | "), " |"))
paste(c(header, sep, rows), collapse = "\n")
}
text_data <- read_csv("data/text_data.csv", show_col_types = FALSE)
expert <- read_csv("data/expert.csv", show_col_types = FALSE)
lr_data <- read_csv("data/lr_data.csv", show_col_types = FALSE)
union_mapping <- read_csv("data/union_mapping.csv", show_col_types = FALSE)
party_families <- read_csv("data/party_families.csv", show_col_types = FALSE)
excluded_poldem <- text_data %>%
filter(project == "PolDem", str_detect(str_to_lower(var), "reform"))
if (nrow(excluded_poldem) > 0) {
stop("Excluded PolDem reform item is present in data/text_data.csv. Final-model diagnostics must be regenerated after removing it.")
}
annual_release <- file.path(release_dir, paste0("party_2d_annual_model_output_", release_version, ".csv.gz"))
panel_release <- file.path(release_dir, paste0("party_2d_election_year_panel_", release_version, ".csv.gz"))
model_positions_file <- latest_file(file.path(outputs_dir, "estimations", "latest"), "^party_positions_.*\\.csv$")
if (is.na(model_positions_file)) {
stop("No post-estimation party-position output found under ", outputs_dir, ". Run model estimation/post-estimation first, or set PARTY2D_OUTPUTS_DIR to an outputs directory.")
}
positions <- read_csv(model_positions_file, show_col_types = FALSE)
item_rows <- bind_rows(
text_data %>%
mutate(source_file = "text_data.csv") %>%
group_by(source_file, item = var, source = project, type_low, type_high) %>%
summarise(observations = n(), party_years = n_distinct(party, year), parties = n_distinct(party), countries = n_distinct(country), min_year = min(year), max_year = max(year), .groups = "drop"),
expert %>%
mutate(source_file = "expert.csv") %>%
group_by(source_file, item = var, source = project, type_low, type_high) %>%
summarise(observations = n(), party_years = n_distinct(party, year), parties = n_distinct(party), countries = n_distinct(country), min_year = min(year), max_year = max(year), .groups = "drop"),
lr_data %>%
mutate(source_file = "lr_data.csv", type_low = NA_character_, type_high = NA_character_) %>%
group_by(source_file, item = var, source = project, type_low, type_high) %>%
summarise(observations = n(), party_years = n_distinct(party, year), parties = n_distinct(party), countries = n_distinct(country), min_year = min(year), max_year = max(year), .groups = "drop")
) %>%
mutate(
dimension = infer_dimension(type_low, type_high),
higher_values_indicate = if_else(is.na(type_high), "source-coded left-right", type_high),
reversed_for_reporting = if_else(is_reversed_for_reporting(type_high), "yes", "no")
) %>%
select(source_file, item, source, dimension, type_low, type_high, higher_values_indicate, reversed_for_reporting, observations, party_years, parties, countries, min_year, max_year) %>%
arrange(source_file, source, dimension, item)
source_coverage <- bind_rows(
text_data %>% transmute(source_file = "text_data.csv", source = project, item = var, party, country, year),
expert %>% transmute(source_file = "expert.csv", source = project, item = var, party, country, year),
lr_data %>% transmute(source_file = "lr_data.csv", source = project, item = var, party, country, year)
) %>%
group_by(source_file, source) %>%
summarise(items = n_distinct(item), observations = n(), party_years = n_distinct(party, year), parties = n_distinct(party), countries = n_distinct(country), min_year = min(year), max_year = max(year), .groups = "drop") %>%
arrange(source_file, source)
party_year_source_coverage <- bind_rows(
text_data %>% distinct(party, country, year) %>% mutate(has_text = TRUE, has_expert = FALSE, has_general_lr = FALSE),
expert %>% distinct(party, country, year) %>% mutate(has_text = FALSE, has_expert = TRUE, has_general_lr = FALSE),
lr_data %>% distinct(party, country, year) %>% mutate(has_text = FALSE, has_expert = FALSE, has_general_lr = TRUE)
) %>%
group_by(party, country, year) %>%
summarise(has_text = any(has_text), has_expert = any(has_expert), has_general_lr = any(has_general_lr), n_source_types = has_text + has_expert + has_general_lr, .groups = "drop") %>%
arrange(year, country, party)
alliance_union_harmonization <- bind_rows(
tibble(metric = "constituent_mappings", category = "all", value = nrow(union_mapping)),
tibble(metric = "unique_union_or_alliance_ids", category = "all", value = n_distinct(union_mapping$manifesto_pf_id)),
tibble(metric = "unique_constituent_party_ids", category = "all", value = n_distinct(union_mapping$expert_pf_id)),
union_mapping %>% count(country, name = "value") %>% transmute(metric = "mappings_by_country", category = country, value),
union_mapping %>% count(status, name = "value") %>% transmute(metric = "mappings_by_status", category = status, value)
)
party_col <- if ("party_id" %in% names(positions)) "party_id" else "party"
party_family_coverage <- positions %>%
transmute(partyfacts_id = .data[[party_col]], country, year) %>%
inner_join(party_families, by = "partyfacts_id") %>%
group_by(family) %>%
summarise(parties = n_distinct(partyfacts_id), party_years = n(), countries = n_distinct(country), min_year = min(year), max_year = max(year), .groups = "drop") %>%
arrange(desc(party_years))
convergence_summary_file <- latest_file(file.path(outputs_dir, "diagnostics"), "^convergence_summary_.*\\.csv$")
convergence_detail_file <- latest_file(file.path(outputs_dir, "diagnostics"), "^convergence_diagnostics_.*\\.csv$")
if (is.na(convergence_summary_file) || is.na(convergence_detail_file)) {
stop("Convergence diagnostics not found under ", outputs_dir, ". Run the model diagnostics before generating the report.")
}
model_convergence_summary <- read_if_exists(convergence_summary_file) %>%
identity()
model_convergence_by_dimension <- read_if_exists(convergence_detail_file) %>%
group_by(dimension) %>%
summarise(parameters = n(), mean_rhat = mean(rhat, na.rm = TRUE), max_rhat = max(rhat, na.rm = TRUE), min_ess_bulk = min(ess_bulk, na.rm = TRUE), mean_ess_bulk = mean(ess_bulk, na.rm = TRUE), .groups = "drop") %>%
mutate(dimension = public_dimension(dimension)) %>%
arrange(dimension)
convergent_summary_file <- latest_file(file.path(outputs_dir, "validation", "latest"), "^convergent_summary_.*\\.csv$")
discriminant_summary_file <- latest_file(file.path(outputs_dir, "validation", "latest"), "^discriminant_summary_.*\\.csv$")
uncertainty_summary_file <- latest_file(file.path(outputs_dir, "validation", "latest"), "^uncertainty_cic_summary_.*\\.csv$")
external_validation_file <- latest_file(file.path(outputs_dir, "validation", "latest"), "^external_validation_.*\\.csv$")
construct_families_file <- latest_file(file.path(outputs_dir, "validation", "latest"), "^construct_families_.*\\.csv$")
construct_unstable_file <- latest_file(file.path(outputs_dir, "validation", "latest"), "^construct_unstable_.*\\.csv$")
if (any(is.na(c(convergent_summary_file, discriminant_summary_file, uncertainty_summary_file, external_validation_file, construct_families_file, construct_unstable_file)))) {
stop("Validation diagnostics not found under ", outputs_dir, ". Run validation before generating the report.")
}
convergent_summary <- read_if_exists(convergent_summary_file) %>%
mutate(diagnostic = "convergent validity", dimension = public_dimension(dimension))
discriminant_summary <- read_if_exists(discriminant_summary_file) %>%
mutate(diagnostic = "discriminant validity", model_dim = public_dimension(model_dim), expert_dim = public_dimension(expert_dim))
uncertainty_summary <- read_if_exists(uncertainty_summary_file) %>%
mutate(diagnostic = "posterior predictive coverage", dimension = public_dimension(dimension))
external_validation_correlations <- read_if_exists(external_validation_file) %>%
group_by(var, dimension) %>%
summarise(n = n(), pearson_r = cor(expert_val, model_val, use = "complete.obs"), mean_absolute_error = mean(abs_error, na.rm = TRUE), coverage_95 = mean(covered_95, na.rm = TRUE), .groups = "drop") %>%
mutate(dimension = public_dimension(dimension)) %>%
arrange(dimension, var)
construct_family_positions <- read_if_exists(construct_families_file) %>%
rename(mean_cultural = mean_galtan, sd_cultural = sd_galtan) %>%
arrange(mean_economic)
construct_temporal_stability <- read_if_exists(construct_unstable_file) %>%
mutate(dimension = public_dimension(dimension)) %>%
arrange(desc(annual_change))
source_composition_balance <- read_if_exists(supplementary_file("validation", "source_composition_balance.csv")) %>%
mutate(dimension = public_dimension(dimension))
robustness_sensitivity <- read_if_exists(supplementary_file("validation", "table10_sensitivity.csv")) %>%
mutate(
dimension = public_dimension(dimension),
across(everything(), ~ na_if(as.character(.x), "[INSERT VALUE]"))
) %>%
select(specification, ablated_source, dimension, matched_n, correlation_with_production,
mean_abs_difference, median_abs_difference, p95_abs_difference,
mean_interval_width_production, mean_interval_width_ablation)
posterior_uncertainty <- positions %>%
summarise(
rows = n(),
parties = n_distinct(.data[[party_col]]),
countries = n_distinct(country),
min_year = min(year),
max_year = max(year),
mean_economic_se = mean(economic_lr_se, na.rm = TRUE),
median_economic_se = median(economic_lr_se, na.rm = TRUE),
mean_cultural_se = mean(galtan_se, na.rm = TRUE),
median_cultural_se = median(galtan_se, na.rm = TRUE)
)
write_csv(item_rows, file.path(generated_dir, "item_coverage.csv"))
write_csv(source_coverage, file.path(generated_dir, "source_coverage.csv"))
write_csv(party_year_source_coverage, file.path(generated_dir, "party_year_source_coverage.csv"))
write_csv(item_rows, file.path(generated_dir, "item_coding_orientation.csv"))
write_csv(filter(item_rows, reversed_for_reporting == "yes"), file.path(generated_dir, "reversed_items.csv"))
write_csv(alliance_union_harmonization, file.path(generated_dir, "alliance_union_harmonization.csv"))
write_csv(party_family_coverage, file.path(generated_dir, "party_family_coverage.csv"))
write_csv(model_convergence_summary, file.path(generated_dir, "model_convergence_summary.csv"))
write_csv(model_convergence_by_dimension, file.path(generated_dir, "model_convergence_by_dimension.csv"))
write_csv(convergent_summary, file.path(generated_dir, "posterior_validation_convergent_summary.csv"))
write_csv(discriminant_summary, file.path(generated_dir, "posterior_validation_discriminant_summary.csv"))
write_csv(uncertainty_summary, file.path(generated_dir, "posterior_validation_uncertainty_summary.csv"))
write_csv(external_validation_correlations, file.path(generated_dir, "external_validation_correlations.csv"))
write_csv(construct_family_positions, file.path(generated_dir, "construct_family_positions.csv"))
write_csv(construct_temporal_stability, file.path(generated_dir, "construct_temporal_stability_flags.csv"))
write_csv(source_composition_balance, file.path(generated_dir, "source_composition_balance.csv"))
write_csv(robustness_sensitivity, file.path(generated_dir, "robustness_sensitivity.csv"))
write_csv(posterior_uncertainty, file.path(generated_dir, "posterior_uncertainty_summary.csv"))
item_counts <- item_rows %>% count(source_file, dimension, name = "items")
source_counts <- source_coverage %>% select(source_file, source, items, observations, party_years, parties, countries, min_year, max_year)
reversed_items <- filter(item_rows, reversed_for_reporting == "yes") %>% select(item, source, dimension, higher_values_indicate, observations, min_year, max_year)
conv_display <- model_convergence_summary %>% select(-any_of("source_file"))
conv_dim_display <- model_convergence_by_dimension %>% select(-any_of("source_file")) %>% mutate(across(where(is.numeric), ~ round(.x, 3)))
val_display <- bind_rows(
convergent_summary %>% transmute(diagnostic, dimension, n, pearson_r = round(r_pearson, 3), spearman_r = round(r_spearman, 3), mae = round(mae, 3), coverage = NA_real_),
uncertainty_summary %>% transmute(diagnostic, dimension, n, pearson_r = NA_real_, spearman_r = NA_real_, mae = NA_real_, coverage = round(cic, 3))
)
report_lines <- c(
"# Diagnostics report",
"",
paste0("Generated: ", format(Sys.time(), "%Y-%m-%d %H:%M:%S %Z")),
paste0("Release: ", release_version),
paste0("Model positions source: `", display_path(model_positions_file), "`"),
"",
"## Purpose",
"",
"The purpose of this report is to provide the appendix-style diagnostics that document how the released party-position estimates are constructed, checked, and validated. The main article reports the central validation evidence; this report keeps the larger technical tables with the release so readers can inspect item coverage, source coverage, coding orientation, harmonization, convergence, posterior uncertainty, and validation details in one reproducible place.",
"",
"## Overview",
"",
"This report follows the structure of the technical appendix material: data and item coverage, coding and scale orientation, party-union harmonization, construct checks, model convergence, and validation. It is generated from the model-ready inputs and completed model outputs; it is not part of the raw-data setup workflow.",
"",
"## Data and item coverage",
"",
"The model combines text-coded item counts, dimension-specific expert placements, and general left-right expert placements. Text items enter as positive/sample counts, expert items enter as aggregated ratings with scale and expert-count information, and general left-right ratings anchor the relationship between the two dimensions.",
"",
md_table(item_counts),
"",
"### Source coverage",
"",
md_table(source_counts),
"",
"## Data coding and item orientation",
"",
"All indicators are oriented toward the two reported dimensions: economic left-right and cultural cosmopolitan--traditionalist. For interpretability, generated diagnostics report whether higher observed values point toward the public high pole or are reversed for reporting. Original source variable names are preserved in the tables.",
"",
"### Reversed items",
"",
md_table(reversed_items),
"",
"## Party unions and electoral coalitions",
"",
"Alliance and union labels are handled through constituent mappings so the released party identifiers represent individual parties. Shared text evidence can inform constituent parties through the union mapping while expert data continue to constrain individual parties directly.",
"",
md_table(alliance_union_harmonization %>% head(30)),
"",
"## Party-family coverage",
"",
"Party-family classifications are used for construct-validity diagnostics and coverage summaries. The table below reports coverage in the completed model output by family code.",
"",
md_table(party_family_coverage),
"",
"### Construct-validity family means",
"",
"Substantive party-family means provide a construct-validity check: families should follow the expected ordering on the economic left-right and cultural cosmopolitan--traditionalist dimensions.",
"",
md_table(construct_family_positions %>% select(family_name, n_parties, n_obs, mean_economic, sd_economic, mean_cultural, sd_cultural) %>% mutate(across(where(is.numeric), ~ round(.x, 3)))),
"",
"### Temporal-stability flags",
"",
"The model permits movement through random walks, but unusually large one-year changes are flagged for inspection rather than treated as automatic errors.",
"",
md_table(construct_temporal_stability %>% select(party_id, country, dimension, year_from, year_to, val_from, val_to, annual_change) %>% mutate(across(where(is.numeric), ~ round(.x, 3))), n = 20),
"",
"## Model convergence diagnostics",
"",
if (nrow(model_convergence_summary) > 0) "Convergence is assessed using split R-hat and effective sample size over monitored parameters." else "Convergence summary files were not found in the configured outputs directory.",
"",
md_table(conv_display),
"",
"### Convergence by parameter group",
"",
md_table(conv_dim_display),
"",
"## Posterior uncertainty",
"",
"The completed party-position output reports posterior standard errors and interval endpoints for both dimensions. These summaries describe the typical uncertainty in the release file used by the report.",
"",
md_table(posterior_uncertainty %>% mutate(across(where(is.numeric), ~ round(.x, 3)))),
"",
"## Validation diagnostics",
"",
"The validation diagnostics combine convergent and discriminant comparisons with expert surveys, posterior predictive coverage, construct checks, and out-of-sample validation when the corresponding outputs are available.",
"",
md_table(val_display),
"",
"### Discriminant validity",
"",
md_table(discriminant_summary %>% select(-any_of("source_file")) %>% mutate(across(where(is.numeric), ~ round(.x, 3)))),
"",
"### External validation correlations",
"",
md_table(external_validation_correlations %>% select(-any_of("source_file")) %>% mutate(across(where(is.numeric), ~ round(.x, 3)))),
"",
"## Evidence-composition balance",
"",
"Evidence-composition balance checks whether estimates informed by different nearby source combinations are systematically shifted relative to rows with both text and expert evidence. The reported differences are adjusted differences on the unit scale relative to the overlapping text-and-expert reference category.",
"",
md_table(source_composition_balance),
"",
"## Robustness and sensitivity checks",
"",
"Sensitivity checks compare the released election-year estimates with source-ablation, segmentation-threshold, and item-screening variants where available. Correlations near one and small absolute differences indicate that the released estimates are stable to the corresponding design choice.",
"",
md_table(robustness_sensitivity),
"",
"## Generated tables",
"",
paste0("- `", list.files(generated_dir, pattern = "\\.csv$"), "`"),
""
)
pdf_source <- file.path(generated_dir, "diagnostics_report.Rmd")
pdf_file <- file.path(generated_dir, "diagnostics_report.pdf")
release_pdf <- file.path(release_dir, paste0("party_2d_diagnostics_report_", release_version, ".pdf"))
pdf_lines <- c(
"---",
"title: \"Diagnostics report\"",
paste0("date: \"", format(Sys.time(), "%Y-%m-%d"), "\""),
"output:",
" pdf_document:",
" toc: true",
" number_sections: true",
" latex_engine: pdflatex",
"geometry: margin=0.75in",
"fontsize: 10pt",
"header-includes:",
" - \\usepackage{booktabs}",
" - \\usepackage{longtable}",
" - \\usepackage{array}",
" - \\usepackage{pdflscape}",
" - \\setlength{\\tabcolsep}{4pt}",
" - \\renewcommand{\\arraystretch}{1.12}",
"---",
"",
"```{r setup, include=FALSE}",
"knitr::opts_chunk$set(echo = FALSE, message = FALSE, warning = FALSE)",
"print_table <- function(x, n = Inf, size = 'footnotesize') {",
" if (nrow(x) == 0) { cat('No rows available.\\n'); return(invisible(NULL)) }",
" x <- head(x, n)",
" x <- mutate(x, across(everything(), as.character))",
" x[is.na(x)] <- ''",
" names(x) <- gsub('_', ' ', names(x), fixed = TRUE)",
" cat(paste0(\"\\n\\\\begingroup\\\\\", size, \"\\n\"))",
" print(knitr::kable(x, format = 'latex', booktabs = TRUE, longtable = FALSE, digits = 3))",
" cat(\"\\n\\\\endgroup\\n\")",
"}",
"short_dim <- function(x) dplyr::recode(as.character(x), 'cultural cosmopolitan--traditionalist' = 'cultural', 'economic left-right' = 'economic', .default = as.character(x))",
"```",
"",
paste0("Generated: ", format(Sys.time(), "%Y-%m-%d %H:%M:%S %Z")),
"",
paste0("Release: ", release_version),
"",
paste0("Model positions source: `", display_path(model_positions_file), "`"),
"",
"# Purpose",
"",
"The purpose of this report is to provide the appendix-style diagnostics that document how the released party-position estimates are constructed, checked, and validated. The main article reports the central validation evidence; this report keeps the larger technical tables with the release so readers can inspect item coverage, source coverage, coding orientation, harmonization, convergence, posterior uncertainty, and validation details in one reproducible place.",
"",
"# Overview",
"",
"This report follows the structure of the technical appendix material: data and item coverage, coding and scale orientation, party-union harmonization, construct checks, model convergence, and validation. It is generated from the model-ready inputs and completed model outputs; it is not part of the raw-data setup workflow.",
"",
"# Data and item coverage",
"",
"The model combines text-coded item counts, dimension-specific expert placements, and general left-right expert placements. Text items enter as positive/sample counts, expert items enter as aggregated ratings with scale and expert-count information, and general left-right ratings anchor the relationship between the two dimensions.",
"",
"```{r item-counts, results='asis'}",
"print_table(item_counts %>% mutate(dimension = short_dim(dimension)))",
"```",
"",
"## Source coverage",
"",
"```{r source-coverage, results='asis'}",
"print_table(source_counts %>% transmute(file = recode(source_file, text_data.csv = 'text', expert.csv = 'expert', lr_data.csv = 'general LR'), source, items, obs = observations, party_years, parties, countries, years = paste0(min_year, '--', max_year)), size = 'scriptsize')",
"```",
"",
"Full item-level coverage and coding-orientation details are provided as generated CSV files listed at the end of this report.",
"",
"# Data coding and item orientation",
"",
"All indicators are oriented toward the two reported dimensions: economic left-right and cultural cosmopolitan--traditionalist. For interpretability, generated diagnostics report whether higher observed values point toward the public high pole or are reversed for reporting. Original source variable names are preserved in the tables.",
"",
"## Reversed items",
"",
"```{r reversed-items, results='asis'}",
"print_table(reversed_items %>% count(source, dimension = short_dim(dimension), higher_values_indicate, name = 'items'))",
"```",
"",
"# Party unions and electoral coalitions",
"",
"Alliance and union labels are handled through constituent mappings so the released party identifiers represent individual parties. Shared text evidence can inform constituent parties through the union mapping while expert data continue to constrain individual parties directly.",
"",
"```{r union-summary, results='asis'}",
"print_table(alliance_union_harmonization %>% transmute(metric, category, value), n = 40)",
"```",
"",
"# Party-family and construct coverage",
"",
"Party-family classifications are used for construct-validity diagnostics and coverage summaries. The table below reports coverage in the completed model output by family code.",
"",
"```{r family-coverage, results='asis'}",
"print_table(party_family_coverage %>% transmute(family, parties, party_years, countries, years = paste0(min_year, '--', max_year)))",
"```",
"",
"## Construct-validity family means",
"",
"Substantive party-family means provide a construct-validity check: families should follow the expected ordering on the economic left-right and cultural cosmopolitan--traditionalist dimensions.",
"",
"```{r construct-family, results='asis'}",
"print_table(construct_family_positions %>% transmute(family = family_name, parties = n_parties, obs = n_obs, econ_mean = round(mean_economic, 3), econ_sd = round(sd_economic, 3), cult_mean = round(mean_cultural, 3), cult_sd = round(sd_cultural, 3)), size = 'scriptsize')",
"```",
"",
"## Temporal-stability flags",
"",
"The model permits movement through random walks, but unusually large one-year changes are flagged for inspection rather than treated as automatic errors.",
"",
"```{r temporal-stability, results='asis'}",
"print_table(construct_temporal_stability %>% transmute(party = party_id, country, dim = short_dim(dimension), from = year_from, to = year_to, start = round(val_from, 3), end = round(val_to, 3), annual_change = round(annual_change, 3)), n = 12, size = 'scriptsize')",
"```",
"",
"# Model convergence diagnostics",
"",
"Convergence is assessed using split R-hat and effective sample size over monitored parameters.",
"",
"```{r convergence-summary, results='asis'}",
"print_table(conv_display)",
"```",
"",
"## Convergence by parameter group",
"",
"```{r convergence-dim, results='asis'}",
"print_table(conv_dim_display %>% mutate(dimension = short_dim(dimension)))",
"```",
"",
"# Posterior uncertainty",
"",
"The completed party-position output reports posterior standard errors and interval endpoints for both dimensions. These summaries describe the typical uncertainty in the release file used by the report.",
"",
"```{r posterior-uncertainty, results='asis'}",
"print_table(posterior_uncertainty %>% transmute(rows, parties, countries, years = paste0(min_year, '--', max_year), mean_econ_se = round(mean_economic_se, 3), median_econ_se = round(median_economic_se, 3), mean_cult_se = round(mean_cultural_se, 3), median_cult_se = round(median_cultural_se, 3)), size = 'scriptsize')",
"```",
"",
"# Validation diagnostics",
"",
"The validation diagnostics combine convergent and discriminant comparisons with expert surveys, posterior predictive coverage, construct checks, and out-of-sample validation when the corresponding outputs are available.",
"",
"```{r validation-summary, results='asis'}",
"print_table(val_display %>% mutate(dimension = short_dim(dimension)), size = 'scriptsize')",
"```",
"",
"## Discriminant validity",
"",
"```{r discriminant, results='asis'}",
"print_table(discriminant_summary %>% transmute(type, model = short_dim(model_dim), expert = expert_dim, n, pearson = round(r_pearson, 3), spearman = round(r_spearman, 3)))",
"```",
"",
"## External validation correlations",
"",
"```{r external-validation, results='asis'}",
"print_table(external_validation_correlations %>% transmute(item = var, dim = short_dim(dimension), n, r = round(pearson_r, 3), mae = round(mean_absolute_error, 3), coverage = round(coverage_95, 3)))",
"```",
"",
"# Evidence-composition balance",
"",
"Evidence-composition balance checks whether estimates informed by different nearby source combinations are systematically shifted relative to rows with both text and expert evidence. The reported differences are adjusted differences on the unit scale relative to the overlapping text-and-expert reference category.",
"",
"```{r source-balance, results='asis'}",
"print_table(source_composition_balance %>% transmute(dim = short_dim(dimension), evidence = recode(source_composition_class, both_direct_or_nearby = 'both', text_only_direct_or_nearby = 'text only', expert_only_direct_or_nearby = 'expert only', temporal_propagation = 'temporal'), ref = recode(reference_class, both_direct_or_nearby = 'both'), n, adj_diff = round(adjusted_difference, 3)))",
"```",
"",
"# Robustness and sensitivity checks",
"",
"Sensitivity checks compare the released election-year estimates with source-ablation, segmentation-threshold, and item-screening variants where available. Correlations near one and small absolute differences indicate that the released estimates are stable to the corresponding design choice.",
"",
"```{r robustness-sensitivity, results='asis'}",
"print_table(robustness_sensitivity %>% transmute(spec = specification, source = ablated_source, dim = short_dim(dimension), n = matched_n, r = round(as.numeric(correlation_with_production), 3), mean_abs = round(as.numeric(mean_abs_difference), 3), median_abs = round(as.numeric(median_abs_difference), 3), p95_abs = round(as.numeric(p95_abs_difference), 3)), size = 'scriptsize')",
"```",
"",
"# Generated tables",
"",
paste0("- `", list.files(generated_dir, pattern = "\\.csv$"), "`")
)
writeLines(pdf_lines, pdf_source)
if (!requireNamespace("rmarkdown", quietly = TRUE)) {
stop("The rmarkdown package is required to render the diagnostics PDF.")
}
rmarkdown::render(
input = pdf_source,
output_format = rmarkdown::pdf_document(toc = TRUE, number_sections = TRUE),
output_file = basename(pdf_file),
output_dir = generated_dir,
quiet = TRUE,
envir = environment()
)
invisible(file.copy(pdf_file, release_pdf, overwrite = TRUE))
unlink(c(
pdf_source,
file.path(generated_dir, "diagnostics_report.log"),
file.path(generated_dir, "diagnostics_report.aux"),
file.path(generated_dir, "diagnostics_report.out"),
file.path(generated_dir, "diagnostics_report.toc"),
file.path(generated_dir, "diagnostics_report.tex")
), force = TRUE)
generated_readme <- c(
"# Generated diagnostics",
"",
"These files are generated by:",
"",
"```bash",
"Rscript diagnostics/generate_diagnostics.R",
"```",
"",
"The command requires completed model/post-estimation outputs. If those outputs are outside the repository root, set `PARTY2D_OUTPUTS_DIR` before running the script.",
"",
"The report file is `diagnostics_report.pdf`; the same PDF is copied into `data/releases/` for the release files."
)
writeLines(generated_readme, file.path(generated_dir, "README.md"))
sha_file <- file.path(release_dir, "SHA256SUMS")
release_files_for_sha <- c(
paste0("party_2d_election_year_panel_", release_version, ".csv.gz"),
paste0("party_2d_annual_model_output_", release_version, ".csv.gz"),
basename(release_pdf)
)
existing_release_files <- release_files_for_sha[file.exists(file.path(release_dir, release_files_for_sha))]
sha_lines <- vapply(existing_release_files, function(f) {
old <- getwd()
on.exit(setwd(old), add = TRUE)
setwd(release_dir)
system2("sha256sum", f, stdout = TRUE)
}, character(1))
writeLines(sha_lines, sha_file)
message("Diagnostics written to diagnostics/generated")
message("Release diagnostics PDF written to ", release_pdf)
+11
View File
@@ -0,0 +1,11 @@
# Generated diagnostics
These files are generated by:
```bash
Rscript diagnostics/generate_diagnostics.R
```
The command requires completed model/post-estimation outputs. If those outputs are outside the repository root, set `PARTY2D_OUTPUTS_DIR` before running the script.
The report file is `diagnostics_report.pdf`; the same PDF is copied into `data/releases/` for the release files.
@@ -0,0 +1,39 @@
metric,category,value
constituent_mappings,all,106
unique_union_or_alliance_ids,all,76
unique_constituent_party_ids,all,100
mappings_by_country,AR,5
mappings_by_country,AU,2
mappings_by_country,BA,1
mappings_by_country,BE,4
mappings_by_country,BG,5
mappings_by_country,BR,6
mappings_by_country,CH,1
mappings_by_country,CL,8
mappings_by_country,CZ,4
mappings_by_country,DE,3
mappings_by_country,EC,1
mappings_by_country,EE,3
mappings_by_country,ES,3
mappings_by_country,FR,4
mappings_by_country,GR,3
mappings_by_country,HR,4
mappings_by_country,IE,1
mappings_by_country,IL,2
mappings_by_country,IT,8
mappings_by_country,LK,3
mappings_by_country,LT,4
mappings_by_country,LU,1
mappings_by_country,LV,6
mappings_by_country,MD,1
mappings_by_country,ME,1
mappings_by_country,MX,2
mappings_by_country,NL,1
mappings_by_country,NZ,1
mappings_by_country,PE,1
mappings_by_country,PL,3
mappings_by_country,RO,6
mappings_by_country,RS,3
mappings_by_country,SK,4
mappings_by_country,UY,1
mappings_by_status,implemented,106
1 metric category value
2 constituent_mappings all 106
3 unique_union_or_alliance_ids all 76
4 unique_constituent_party_ids all 100
5 mappings_by_country AR 5
6 mappings_by_country AU 2
7 mappings_by_country BA 1
8 mappings_by_country BE 4
9 mappings_by_country BG 5
10 mappings_by_country BR 6
11 mappings_by_country CH 1
12 mappings_by_country CL 8
13 mappings_by_country CZ 4
14 mappings_by_country DE 3
15 mappings_by_country EC 1
16 mappings_by_country EE 3
17 mappings_by_country ES 3
18 mappings_by_country FR 4
19 mappings_by_country GR 3
20 mappings_by_country HR 4
21 mappings_by_country IE 1
22 mappings_by_country IL 2
23 mappings_by_country IT 8
24 mappings_by_country LK 3
25 mappings_by_country LT 4
26 mappings_by_country LU 1
27 mappings_by_country LV 6
28 mappings_by_country MD 1
29 mappings_by_country ME 1
30 mappings_by_country MX 2
31 mappings_by_country NL 1
32 mappings_by_country NZ 1
33 mappings_by_country PE 1
34 mappings_by_country PL 3
35 mappings_by_country RO 6
36 mappings_by_country RS 3
37 mappings_by_country SK 4
38 mappings_by_country UY 1
39 mappings_by_status implemented 106
@@ -0,0 +1,8 @@
family,n_parties,n_obs,mean_economic,sd_economic,mean_cultural,sd_cultural,family_name
com,49,1241,0.12001647469327872,0.0901861773223108,0.3834027613298184,0.18041637351915937,Communist/Far Left
eco,30,723,0.26720980626115975,0.11467133878653364,0.2514849926574344,0.09223964864606182,Green/Ecological
soc,86,2892,0.3305567447692131,0.1240506242334158,0.3777170994931328,0.13783788204153472,Social Democratic
chr,41,1596,0.5961213268671609,0.11930537492981286,0.5293978691891922,0.13964161332639527,Christian Democratic
right,50,975,0.6222552715406671,0.18991001824381296,0.720426765976975,0.15147355617267264,Radical Right
con,82,2425,0.6519453485719768,0.17791542785462533,0.5375091531921928,0.14258881749137706,Conservative
lib,81,2063,0.6569946895012412,0.15224833699588255,0.37069062594934565,0.13158056834168683,Liberal
1 family n_parties n_obs mean_economic sd_economic mean_cultural sd_cultural family_name
2 com 49 1241 0.12001647469327872 0.0901861773223108 0.3834027613298184 0.18041637351915937 Communist/Far Left
3 eco 30 723 0.26720980626115975 0.11467133878653364 0.2514849926574344 0.09223964864606182 Green/Ecological
4 soc 86 2892 0.3305567447692131 0.1240506242334158 0.3777170994931328 0.13783788204153472 Social Democratic
5 chr 41 1596 0.5961213268671609 0.11930537492981286 0.5293978691891922 0.13964161332639527 Christian Democratic
6 right 50 975 0.6222552715406671 0.18991001824381296 0.720426765976975 0.15147355617267264 Radical Right
7 con 82 2425 0.6519453485719768 0.17791542785462533 0.5375091531921928 0.14258881749137706 Conservative
8 lib 81 2063 0.6569946895012412 0.15224833699588255 0.37069062594934565 0.13158056834168683 Liberal
@@ -0,0 +1,60 @@
party_id,country,dimension,year_from,year_to,val_from,val_to,change,annual_change
556,LT,cultural cosmopolitan--traditionalist,2019,2020,0.77107951125,0.5289456789999999,0.24213383225000007,0.24213383225000007
1663,IL,cultural cosmopolitan--traditionalist,2021,2022,0.1423029388125,0.35615504375,0.2138521049375,0.2138521049375
455,IL,cultural cosmopolitan--traditionalist,1997,1998,0.456850958125,0.6432782493750001,0.18642729125000007,0.18642729125000007
455,IL,cultural cosmopolitan--traditionalist,1996,1997,0.27663313025,0.456850958125,0.180217827875,0.180217827875
8393,LV,cultural cosmopolitan--traditionalist,2018,2019,0.4386382122500001,0.2627366591625,0.17590155308750005,0.17590155308750005
281,BE,cultural cosmopolitan--traditionalist,1977,1978,0.6706878695,0.843416257375,0.172728387875,0.172728387875
556,LT,cultural cosmopolitan--traditionalist,2001,2002,0.318926108375,0.49125159325,0.172325484875,0.172325484875
964,IS,economic left-right,2016,2017,0.691268056,0.520838928125,0.17042912787499995,0.17042912787499995
298,NL,cultural cosmopolitan--traditionalist,2018,2019,0.760919900625,0.592874692125,0.16804520850000004,0.16804520850000004
901,FI,economic left-right,1992,1993,0.486145510375,0.64721287575,0.161067365375,0.161067365375
467,SI,cultural cosmopolitan--traditionalist,2018,2019,0.563748886625,0.4041139849999999,0.15963490162500005,0.15963490162500005
1221,IT,economic left-right,2007,2008,0.559621162375,0.400493563,0.15912759937500004,0.15912759937500004
901,FI,economic left-right,1991,1992,0.327277968125,0.486145510375,0.15886754225,0.15886754225
455,IL,cultural cosmopolitan--traditionalist,1998,1999,0.6432782493750001,0.7970417803750001,0.15376353099999995,0.15376353099999995
1221,IT,economic left-right,2006,2007,0.7086087693750001,0.559621162375,0.14898760700000002,0.14898760700000002
2211,UA,economic left-right,2006,2007,0.416653024,0.5619799204999999,0.1453268964999999,0.1453268964999999
631,CH,economic left-right,2018,2019,0.613235049875,0.7569515025,0.143716452625,0.143716452625
298,NL,cultural cosmopolitan--traditionalist,2019,2020,0.592874692125,0.734536642,0.14166194987500005,0.14166194987500005
556,LT,cultural cosmopolitan--traditionalist,2000,2001,0.180505337875,0.318926108375,0.1384207705,0.1384207705
1417,IL,cultural cosmopolitan--traditionalist,1968,1969,0.49841048887499995,0.635275993875,0.13686550500000003,0.13686550500000003
81,ES,cultural cosmopolitan--traditionalist,1999,2000,0.44213168437499994,0.30819645050000005,0.1339352338749999,0.1339352338749999
1417,IL,cultural cosmopolitan--traditionalist,1967,1968,0.364764852,0.49841048887499995,0.13364563687499997,0.13364563687499997
901,FI,economic left-right,1993,1994,0.64721287575,0.78024709825,0.13303422250000008,0.13303422250000008
409,SE,economic left-right,2018,2019,0.4993397851250001,0.6246204093750001,0.12528062425000003,0.12528062425000003
48,GR,cultural cosmopolitan--traditionalist,1999,2000,0.471858772375,0.594323158,0.122464385625,0.122464385625
212,DK,cultural cosmopolitan--traditionalist,2014,2015,0.339423171125,0.459806116125,0.12038294500000002,0.12038294500000002
2415,IT,cultural cosmopolitan--traditionalist,2006,2007,0.6143681051250001,0.49546903375,0.11889907137500004,0.11889907137500004
1369,IT,cultural cosmopolitan--traditionalist,2013,2014,0.7406959332499999,0.6231419237500002,0.11755400949999972,0.11755400949999972
5852,IS,cultural cosmopolitan--traditionalist,2018,2019,0.336059923125,0.45280653625,0.116746613125,0.116746613125
1417,IL,cultural cosmopolitan--traditionalist,1966,1967,0.2487840847,0.364764852,0.11598076729999995,0.11598076729999995
828,NL,cultural cosmopolitan--traditionalist,2019,2020,0.399398152875,0.5144400794999999,0.11504192662499996,0.11504192662499996
2415,IT,cultural cosmopolitan--traditionalist,2007,2008,0.49546903375,0.3804985986625,0.1149704350875,0.1149704350875
828,NL,cultural cosmopolitan--traditionalist,2020,2021,0.5144400794999999,0.6280684987500001,0.11362841925000022,0.11362841925000022
573,DE,cultural cosmopolitan--traditionalist,2024,2025,0.34241723825000003,0.4558404575,0.11342321924999998,0.11342321924999998
1424,BE,economic left-right,1977,1978,0.7125746831249999,0.825810219125,0.11323553600000004,0.11323553600000004
1173,NO,cultural cosmopolitan--traditionalist,2018,2019,0.353365422125,0.46349993075,0.11013450862499996,0.11013450862499996
1651,GR,economic left-right,2013,2014,0.441388039625,0.551427035875,0.11003899624999997,0.11003899624999997
1660,GR,cultural cosmopolitan--traditionalist,2012,2013,0.70537148075,0.8146704603749999,0.1092989796249999,0.1092989796249999
433,FR,economic left-right,2018,2019,0.433590134625,0.542142367875,0.10855223325000002,0.10855223325000002
1002,GB,cultural cosmopolitan--traditionalist,2014,2015,0.3184758865,0.2101810030625,0.10829488343750002,0.10829488343750002
623,AR,economic left-right,1989,1990,0.4145806991249999,0.5228341057499999,0.10825340662499994,0.10825340662499994
1305,RO,economic left-right,2000,2001,0.553705481625,0.446037918375,0.10766756325,0.10766756325
298,NL,cultural cosmopolitan--traditionalist,2020,2021,0.734536642,0.84215486075,0.10761821875,0.10761821875
1359,PT,economic left-right,2004,2005,0.495169041875,0.602184326875,0.10701528500000002,0.10701528500000002
1651,GR,economic left-right,2012,2013,0.3355888995,0.441388039625,0.10579914012500002,0.10579914012500002
623,AR,economic left-right,1990,1991,0.5228341057499999,0.62815575375,0.1053216480000001,0.1053216480000001
1305,RO,economic left-right,2001,2002,0.446037918375,0.3412955855,0.104742332875,0.104742332875
455,IL,cultural cosmopolitan--traditionalist,1991,1992,0.5368312063749999,0.432172542875,0.10465866349999992,0.10465866349999992
2415,IT,cultural cosmopolitan--traditionalist,2008,2009,0.3804985986625,0.2762634036875,0.104235194975,0.104235194975
599,AT,economic left-right,2008,2009,0.4633838822500001,0.567595171125,0.10421128887499996,0.10421128887499996
669,CH,cultural cosmopolitan--traditionalist,2016,2017,0.514819538625,0.410640874875,0.10417866374999996,0.10417866374999996
5852,IS,cultural cosmopolitan--traditionalist,2017,2018,0.23270289265,0.336059923125,0.10335703047499996,0.10335703047499996
669,CH,cultural cosmopolitan--traditionalist,2015,2016,0.617986882875,0.514819538625,0.10316734425000008,0.10316734425000008
48,GR,cultural cosmopolitan--traditionalist,2010,2011,0.569647428,0.6723034049999999,0.10265597699999984,0.10265597699999984
1221,IT,economic left-right,2008,2009,0.400493563,0.50303734575,0.10254378275000003,0.10254378275000003
1651,GR,economic left-right,2014,2015,0.551427035875,0.6535508147500001,0.10212377887500013,0.10212377887500013
1221,IT,economic left-right,2009,2010,0.50303734575,0.604409204875,0.10137185912500002,0.10137185912500002
975,SI,economic left-right,1990,1991,0.579305296625,0.6803467895000002,0.1010414928750002,0.1010414928750002
338,AU,economic left-right,1992,1993,0.791994626125,0.6916490538750002,0.10034557224999983,0.10034557224999983
1 party_id country dimension year_from year_to val_from val_to change annual_change
2 556 LT cultural cosmopolitan--traditionalist 2019 2020 0.77107951125 0.5289456789999999 0.24213383225000007 0.24213383225000007
3 1663 IL cultural cosmopolitan--traditionalist 2021 2022 0.1423029388125 0.35615504375 0.2138521049375 0.2138521049375
4 455 IL cultural cosmopolitan--traditionalist 1997 1998 0.456850958125 0.6432782493750001 0.18642729125000007 0.18642729125000007
5 455 IL cultural cosmopolitan--traditionalist 1996 1997 0.27663313025 0.456850958125 0.180217827875 0.180217827875
6 8393 LV cultural cosmopolitan--traditionalist 2018 2019 0.4386382122500001 0.2627366591625 0.17590155308750005 0.17590155308750005
7 281 BE cultural cosmopolitan--traditionalist 1977 1978 0.6706878695 0.843416257375 0.172728387875 0.172728387875
8 556 LT cultural cosmopolitan--traditionalist 2001 2002 0.318926108375 0.49125159325 0.172325484875 0.172325484875
9 964 IS economic left-right 2016 2017 0.691268056 0.520838928125 0.17042912787499995 0.17042912787499995
10 298 NL cultural cosmopolitan--traditionalist 2018 2019 0.760919900625 0.592874692125 0.16804520850000004 0.16804520850000004
11 901 FI economic left-right 1992 1993 0.486145510375 0.64721287575 0.161067365375 0.161067365375
12 467 SI cultural cosmopolitan--traditionalist 2018 2019 0.563748886625 0.4041139849999999 0.15963490162500005 0.15963490162500005
13 1221 IT economic left-right 2007 2008 0.559621162375 0.400493563 0.15912759937500004 0.15912759937500004
14 901 FI economic left-right 1991 1992 0.327277968125 0.486145510375 0.15886754225 0.15886754225
15 455 IL cultural cosmopolitan--traditionalist 1998 1999 0.6432782493750001 0.7970417803750001 0.15376353099999995 0.15376353099999995
16 1221 IT economic left-right 2006 2007 0.7086087693750001 0.559621162375 0.14898760700000002 0.14898760700000002
17 2211 UA economic left-right 2006 2007 0.416653024 0.5619799204999999 0.1453268964999999 0.1453268964999999
18 631 CH economic left-right 2018 2019 0.613235049875 0.7569515025 0.143716452625 0.143716452625
19 298 NL cultural cosmopolitan--traditionalist 2019 2020 0.592874692125 0.734536642 0.14166194987500005 0.14166194987500005
20 556 LT cultural cosmopolitan--traditionalist 2000 2001 0.180505337875 0.318926108375 0.1384207705 0.1384207705
21 1417 IL cultural cosmopolitan--traditionalist 1968 1969 0.49841048887499995 0.635275993875 0.13686550500000003 0.13686550500000003
22 81 ES cultural cosmopolitan--traditionalist 1999 2000 0.44213168437499994 0.30819645050000005 0.1339352338749999 0.1339352338749999
23 1417 IL cultural cosmopolitan--traditionalist 1967 1968 0.364764852 0.49841048887499995 0.13364563687499997 0.13364563687499997
24 901 FI economic left-right 1993 1994 0.64721287575 0.78024709825 0.13303422250000008 0.13303422250000008
25 409 SE economic left-right 2018 2019 0.4993397851250001 0.6246204093750001 0.12528062425000003 0.12528062425000003
26 48 GR cultural cosmopolitan--traditionalist 1999 2000 0.471858772375 0.594323158 0.122464385625 0.122464385625
27 212 DK cultural cosmopolitan--traditionalist 2014 2015 0.339423171125 0.459806116125 0.12038294500000002 0.12038294500000002
28 2415 IT cultural cosmopolitan--traditionalist 2006 2007 0.6143681051250001 0.49546903375 0.11889907137500004 0.11889907137500004
29 1369 IT cultural cosmopolitan--traditionalist 2013 2014 0.7406959332499999 0.6231419237500002 0.11755400949999972 0.11755400949999972
30 5852 IS cultural cosmopolitan--traditionalist 2018 2019 0.336059923125 0.45280653625 0.116746613125 0.116746613125
31 1417 IL cultural cosmopolitan--traditionalist 1966 1967 0.2487840847 0.364764852 0.11598076729999995 0.11598076729999995
32 828 NL cultural cosmopolitan--traditionalist 2019 2020 0.399398152875 0.5144400794999999 0.11504192662499996 0.11504192662499996
33 2415 IT cultural cosmopolitan--traditionalist 2007 2008 0.49546903375 0.3804985986625 0.1149704350875 0.1149704350875
34 828 NL cultural cosmopolitan--traditionalist 2020 2021 0.5144400794999999 0.6280684987500001 0.11362841925000022 0.11362841925000022
35 573 DE cultural cosmopolitan--traditionalist 2024 2025 0.34241723825000003 0.4558404575 0.11342321924999998 0.11342321924999998
36 1424 BE economic left-right 1977 1978 0.7125746831249999 0.825810219125 0.11323553600000004 0.11323553600000004
37 1173 NO cultural cosmopolitan--traditionalist 2018 2019 0.353365422125 0.46349993075 0.11013450862499996 0.11013450862499996
38 1651 GR economic left-right 2013 2014 0.441388039625 0.551427035875 0.11003899624999997 0.11003899624999997
39 1660 GR cultural cosmopolitan--traditionalist 2012 2013 0.70537148075 0.8146704603749999 0.1092989796249999 0.1092989796249999
40 433 FR economic left-right 2018 2019 0.433590134625 0.542142367875 0.10855223325000002 0.10855223325000002
41 1002 GB cultural cosmopolitan--traditionalist 2014 2015 0.3184758865 0.2101810030625 0.10829488343750002 0.10829488343750002
42 623 AR economic left-right 1989 1990 0.4145806991249999 0.5228341057499999 0.10825340662499994 0.10825340662499994
43 1305 RO economic left-right 2000 2001 0.553705481625 0.446037918375 0.10766756325 0.10766756325
44 298 NL cultural cosmopolitan--traditionalist 2020 2021 0.734536642 0.84215486075 0.10761821875 0.10761821875
45 1359 PT economic left-right 2004 2005 0.495169041875 0.602184326875 0.10701528500000002 0.10701528500000002
46 1651 GR economic left-right 2012 2013 0.3355888995 0.441388039625 0.10579914012500002 0.10579914012500002
47 623 AR economic left-right 1990 1991 0.5228341057499999 0.62815575375 0.1053216480000001 0.1053216480000001
48 1305 RO economic left-right 2001 2002 0.446037918375 0.3412955855 0.104742332875 0.104742332875
49 455 IL cultural cosmopolitan--traditionalist 1991 1992 0.5368312063749999 0.432172542875 0.10465866349999992 0.10465866349999992
50 2415 IT cultural cosmopolitan--traditionalist 2008 2009 0.3804985986625 0.2762634036875 0.104235194975 0.104235194975
51 599 AT economic left-right 2008 2009 0.4633838822500001 0.567595171125 0.10421128887499996 0.10421128887499996
52 669 CH cultural cosmopolitan--traditionalist 2016 2017 0.514819538625 0.410640874875 0.10417866374999996 0.10417866374999996
53 5852 IS cultural cosmopolitan--traditionalist 2017 2018 0.23270289265 0.336059923125 0.10335703047499996 0.10335703047499996
54 669 CH cultural cosmopolitan--traditionalist 2015 2016 0.617986882875 0.514819538625 0.10316734425000008 0.10316734425000008
55 48 GR cultural cosmopolitan--traditionalist 2010 2011 0.569647428 0.6723034049999999 0.10265597699999984 0.10265597699999984
56 1221 IT economic left-right 2008 2009 0.400493563 0.50303734575 0.10254378275000003 0.10254378275000003
57 1651 GR economic left-right 2014 2015 0.551427035875 0.6535508147500001 0.10212377887500013 0.10212377887500013
58 1221 IT economic left-right 2009 2010 0.50303734575 0.604409204875 0.10137185912500002 0.10137185912500002
59 975 SI economic left-right 1990 1991 0.579305296625 0.6803467895000002 0.1010414928750002 0.1010414928750002
60 338 AU economic left-right 1992 1993 0.791994626125 0.6916490538750002 0.10034557224999983 0.10034557224999983
Binary file not shown.
@@ -0,0 +1,11 @@
var,dimension,n,pearson_r,mean_absolute_error,coverage_95
culsup_vparty,cultural cosmopolitan--traditionalist,536,0.8121247297636631,0.12821713597308768,0.3843283582089552
galtan_ches,cultural cosmopolitan--traditionalist,222,0.9588005926603757,0.07875607868037135,0.5225225225225225
gender_vparty,cultural cosmopolitan--traditionalist,545,0.5626122704047269,0.16947520363543578,0.28990825688073396
immig_vparty,cultural cosmopolitan--traditionalist,537,0.7429394940511392,0.10311559715251396,0.4897579143389199
lgbt_vparty,cultural cosmopolitan--traditionalist,541,0.7941178030023652,0.09410224229993068,0.5508317929759704
relig_vparty,cultural cosmopolitan--traditionalist,548,0.6757229503671286,0.30309927660661495,0.04744525547445255
lrecon_ches,economic left-right,223,0.9739626905522167,0.05518814853885153,0.8116591928251121
lrecon_poppa,economic left-right,74,0.9799670973969279,0.0660246477855859,0.6621621621621622
lrecon_vparty,economic left-right,534,0.8664105550524236,0.08828332773956499,0.6741573033707865
welf_vparty,economic left-right,534,0.6821895613302613,0.17587920065205523,0.36329588014981273
1 var dimension n pearson_r mean_absolute_error coverage_95
2 culsup_vparty cultural cosmopolitan--traditionalist 536 0.8121247297636631 0.12821713597308768 0.3843283582089552
3 galtan_ches cultural cosmopolitan--traditionalist 222 0.9588005926603757 0.07875607868037135 0.5225225225225225
4 gender_vparty cultural cosmopolitan--traditionalist 545 0.5626122704047269 0.16947520363543578 0.28990825688073396
5 immig_vparty cultural cosmopolitan--traditionalist 537 0.7429394940511392 0.10311559715251396 0.4897579143389199
6 lgbt_vparty cultural cosmopolitan--traditionalist 541 0.7941178030023652 0.09410224229993068 0.5508317929759704
7 relig_vparty cultural cosmopolitan--traditionalist 548 0.6757229503671286 0.30309927660661495 0.04744525547445255
8 lrecon_ches economic left-right 223 0.9739626905522167 0.05518814853885153 0.8116591928251121
9 lrecon_poppa economic left-right 74 0.9799670973969279 0.0660246477855859 0.6621621621621622
10 lrecon_vparty economic left-right 534 0.8664105550524236 0.08828332773956499 0.6741573033707865
11 welf_vparty economic left-right 534 0.6821895613302613 0.17587920065205523 0.36329588014981273
@@ -0,0 +1,33 @@
source_file,item,source,dimension,type_low,type_high,higher_values_indicate,reversed_for_reporting,observations,party_years,parties,countries,min_year,max_year
expert.csv,galtan_ches,CHES,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,1319,1319,389,44,1999,2024
expert.csv,lrecon_ches,CHES,economic left-right,pro_welfare,pro_market,pro_market,no,1320,1320,390,44,1999,2024
expert.csv,libcon_gps,GPS,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,269,269,269,61,2019,2019
expert.csv,lrecon_gps,GPS,economic left-right,pro_welfare,pro_market,pro_market,no,269,269,269,62,2019,2019
expert.csv,lrecon_poppa,POPPA,economic left-right,pro_welfare,pro_market,pro_market,no,413,413,225,31,2018,2023
expert.csv,culsup_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3076,3076,589,65,1970,2019
expert.csv,gender_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3043,3043,586,65,1970,2019
expert.csv,immig_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3076,3076,589,65,1970,2019
expert.csv,lgbt_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3076,3076,589,65,1970,2019
expert.csv,relig_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3076,3076,589,65,1970,2019
expert.csv,lrecon_vparty,V-Party,economic left-right,pro_welfare,pro_market,pro_market,no,3075,3075,588,65,1970,2019
expert.csv,welf_vparty,V-Party,economic left-right,pro_welfare,pro_market,pro_market,no,3066,3066,585,65,1970,2019
lr_data.csv,lr_ches,CHES,general left-right,NA,NA,source-coded left-right,no,1320,1320,390,44,1999,2024
lr_data.csv,lr_morgan,Morgan,general left-right,NA,NA,source-coded left-right,no,471,471,72,11,1945,1973
lr_data.csv,lr_poppa,POPPA,general left-right,NA,NA,source-coded left-right,no,416,416,225,31,2018,2023
text_data.csv,conservative_morality_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,4501,4501,713,65,1920,2025
text_data.csv,internationalism_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,4501,4501,713,65,1920,2025
text_data.csv,multiculturalism_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,4501,4501,713,65,1920,2025
text_data.csv,national_identity_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,4501,4501,713,65,1920,2025
text_data.csv,economic_intervention_manifesto,Manifesto Project,economic left-right,pro_market,pro_welfare,pro_welfare,yes,4501,4501,713,65,1920,2025
text_data.csv,economic_liberalization_manifesto,Manifesto Project,economic left-right,pro_welfare,pro_market,pro_market,no,4501,4501,713,65,1920,2025
text_data.csv,market_regulation_manifesto,Manifesto Project,economic left-right,pro_welfare,pro_market,pro_market,no,4501,4501,713,65,1920,2025
text_data.csv,social_services_manifesto,Manifesto Project,economic left-right,pro_market,pro_welfare,pro_welfare,yes,4501,4501,713,65,1920,2025
text_data.csv,cultlib_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,299,299,78,15,1972,2017
text_data.csv,defense_poldem,PolDem,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,243,243,67,15,1972,2017
text_data.csv,euro_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,93,93,44,13,1978,2017
text_data.csv,europe_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,217,217,66,15,1972,2017
text_data.csv,immig_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,236,236,67,14,1972,2017
text_data.csv,nationalism_poldem,PolDem,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,131,131,60,15,1974,2017
text_data.csv,security_poldem,PolDem,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,265,265,78,15,1972,2017
text_data.csv,ecolib_poldem,PolDem,economic left-right,pro_welfare,pro_market,pro_market,no,361,361,85,15,1972,2017
text_data.csv,welfare_poldem,PolDem,economic left-right,pro_market,pro_welfare,pro_welfare,yes,349,349,83,15,1972,2017
1 source_file item source dimension type_low type_high higher_values_indicate reversed_for_reporting observations party_years parties countries min_year max_year
2 expert.csv galtan_ches CHES cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 1319 1319 389 44 1999 2024
3 expert.csv lrecon_ches CHES economic left-right pro_welfare pro_market pro_market no 1320 1320 390 44 1999 2024
4 expert.csv libcon_gps GPS cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 269 269 269 61 2019 2019
5 expert.csv lrecon_gps GPS economic left-right pro_welfare pro_market pro_market no 269 269 269 62 2019 2019
6 expert.csv lrecon_poppa POPPA economic left-right pro_welfare pro_market pro_market no 413 413 225 31 2018 2023
7 expert.csv culsup_vparty V-Party cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 3076 3076 589 65 1970 2019
8 expert.csv gender_vparty V-Party cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 3043 3043 586 65 1970 2019
9 expert.csv immig_vparty V-Party cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 3076 3076 589 65 1970 2019
10 expert.csv lgbt_vparty V-Party cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 3076 3076 589 65 1970 2019
11 expert.csv relig_vparty V-Party cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 3076 3076 589 65 1970 2019
12 expert.csv lrecon_vparty V-Party economic left-right pro_welfare pro_market pro_market no 3075 3075 588 65 1970 2019
13 expert.csv welf_vparty V-Party economic left-right pro_welfare pro_market pro_market no 3066 3066 585 65 1970 2019
14 lr_data.csv lr_ches CHES general left-right NA NA source-coded left-right no 1320 1320 390 44 1999 2024
15 lr_data.csv lr_morgan Morgan general left-right NA NA source-coded left-right no 471 471 72 11 1945 1973
16 lr_data.csv lr_poppa POPPA general left-right NA NA source-coded left-right no 416 416 225 31 2018 2023
17 text_data.csv conservative_morality_manifesto Manifesto Project cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 4501 4501 713 65 1920 2025
18 text_data.csv internationalism_manifesto Manifesto Project cultural cosmopolitan--traditionalist traditional cosmopolitan cosmopolitan yes 4501 4501 713 65 1920 2025
19 text_data.csv multiculturalism_manifesto Manifesto Project cultural cosmopolitan--traditionalist traditional cosmopolitan cosmopolitan yes 4501 4501 713 65 1920 2025
20 text_data.csv national_identity_manifesto Manifesto Project cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 4501 4501 713 65 1920 2025
21 text_data.csv economic_intervention_manifesto Manifesto Project economic left-right pro_market pro_welfare pro_welfare yes 4501 4501 713 65 1920 2025
22 text_data.csv economic_liberalization_manifesto Manifesto Project economic left-right pro_welfare pro_market pro_market no 4501 4501 713 65 1920 2025
23 text_data.csv market_regulation_manifesto Manifesto Project economic left-right pro_welfare pro_market pro_market no 4501 4501 713 65 1920 2025
24 text_data.csv social_services_manifesto Manifesto Project economic left-right pro_market pro_welfare pro_welfare yes 4501 4501 713 65 1920 2025
25 text_data.csv cultlib_poldem PolDem cultural cosmopolitan--traditionalist traditional cosmopolitan cosmopolitan yes 299 299 78 15 1972 2017
26 text_data.csv defense_poldem PolDem cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 243 243 67 15 1972 2017
27 text_data.csv euro_poldem PolDem cultural cosmopolitan--traditionalist traditional cosmopolitan cosmopolitan yes 93 93 44 13 1978 2017
28 text_data.csv europe_poldem PolDem cultural cosmopolitan--traditionalist traditional cosmopolitan cosmopolitan yes 217 217 66 15 1972 2017
29 text_data.csv immig_poldem PolDem cultural cosmopolitan--traditionalist traditional cosmopolitan cosmopolitan yes 236 236 67 14 1972 2017
30 text_data.csv nationalism_poldem PolDem cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 131 131 60 15 1974 2017
31 text_data.csv security_poldem PolDem cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 265 265 78 15 1972 2017
32 text_data.csv ecolib_poldem PolDem economic left-right pro_welfare pro_market pro_market no 361 361 85 15 1972 2017
33 text_data.csv welfare_poldem PolDem economic left-right pro_market pro_welfare pro_welfare yes 349 349 83 15 1972 2017
+33
View File
@@ -0,0 +1,33 @@
source_file,item,source,dimension,type_low,type_high,higher_values_indicate,reversed_for_reporting,observations,party_years,parties,countries,min_year,max_year
expert.csv,galtan_ches,CHES,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,1319,1319,389,44,1999,2024
expert.csv,lrecon_ches,CHES,economic left-right,pro_welfare,pro_market,pro_market,no,1320,1320,390,44,1999,2024
expert.csv,libcon_gps,GPS,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,269,269,269,61,2019,2019
expert.csv,lrecon_gps,GPS,economic left-right,pro_welfare,pro_market,pro_market,no,269,269,269,62,2019,2019
expert.csv,lrecon_poppa,POPPA,economic left-right,pro_welfare,pro_market,pro_market,no,413,413,225,31,2018,2023
expert.csv,culsup_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3076,3076,589,65,1970,2019
expert.csv,gender_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3043,3043,586,65,1970,2019
expert.csv,immig_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3076,3076,589,65,1970,2019
expert.csv,lgbt_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3076,3076,589,65,1970,2019
expert.csv,relig_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3076,3076,589,65,1970,2019
expert.csv,lrecon_vparty,V-Party,economic left-right,pro_welfare,pro_market,pro_market,no,3075,3075,588,65,1970,2019
expert.csv,welf_vparty,V-Party,economic left-right,pro_welfare,pro_market,pro_market,no,3066,3066,585,65,1970,2019
lr_data.csv,lr_ches,CHES,general left-right,NA,NA,source-coded left-right,no,1320,1320,390,44,1999,2024
lr_data.csv,lr_morgan,Morgan,general left-right,NA,NA,source-coded left-right,no,471,471,72,11,1945,1973
lr_data.csv,lr_poppa,POPPA,general left-right,NA,NA,source-coded left-right,no,416,416,225,31,2018,2023
text_data.csv,conservative_morality_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,4501,4501,713,65,1920,2025
text_data.csv,internationalism_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,4501,4501,713,65,1920,2025
text_data.csv,multiculturalism_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,4501,4501,713,65,1920,2025
text_data.csv,national_identity_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,4501,4501,713,65,1920,2025
text_data.csv,economic_intervention_manifesto,Manifesto Project,economic left-right,pro_market,pro_welfare,pro_welfare,yes,4501,4501,713,65,1920,2025
text_data.csv,economic_liberalization_manifesto,Manifesto Project,economic left-right,pro_welfare,pro_market,pro_market,no,4501,4501,713,65,1920,2025
text_data.csv,market_regulation_manifesto,Manifesto Project,economic left-right,pro_welfare,pro_market,pro_market,no,4501,4501,713,65,1920,2025
text_data.csv,social_services_manifesto,Manifesto Project,economic left-right,pro_market,pro_welfare,pro_welfare,yes,4501,4501,713,65,1920,2025
text_data.csv,cultlib_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,299,299,78,15,1972,2017
text_data.csv,defense_poldem,PolDem,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,243,243,67,15,1972,2017
text_data.csv,euro_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,93,93,44,13,1978,2017
text_data.csv,europe_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,217,217,66,15,1972,2017
text_data.csv,immig_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,236,236,67,14,1972,2017
text_data.csv,nationalism_poldem,PolDem,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,131,131,60,15,1974,2017
text_data.csv,security_poldem,PolDem,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,265,265,78,15,1972,2017
text_data.csv,ecolib_poldem,PolDem,economic left-right,pro_welfare,pro_market,pro_market,no,361,361,85,15,1972,2017
text_data.csv,welfare_poldem,PolDem,economic left-right,pro_market,pro_welfare,pro_welfare,yes,349,349,83,15,1972,2017
1 source_file item source dimension type_low type_high higher_values_indicate reversed_for_reporting observations party_years parties countries min_year max_year
2 expert.csv galtan_ches CHES cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 1319 1319 389 44 1999 2024
3 expert.csv lrecon_ches CHES economic left-right pro_welfare pro_market pro_market no 1320 1320 390 44 1999 2024
4 expert.csv libcon_gps GPS cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 269 269 269 61 2019 2019
5 expert.csv lrecon_gps GPS economic left-right pro_welfare pro_market pro_market no 269 269 269 62 2019 2019
6 expert.csv lrecon_poppa POPPA economic left-right pro_welfare pro_market pro_market no 413 413 225 31 2018 2023
7 expert.csv culsup_vparty V-Party cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 3076 3076 589 65 1970 2019
8 expert.csv gender_vparty V-Party cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 3043 3043 586 65 1970 2019
9 expert.csv immig_vparty V-Party cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 3076 3076 589 65 1970 2019
10 expert.csv lgbt_vparty V-Party cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 3076 3076 589 65 1970 2019
11 expert.csv relig_vparty V-Party cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 3076 3076 589 65 1970 2019
12 expert.csv lrecon_vparty V-Party economic left-right pro_welfare pro_market pro_market no 3075 3075 588 65 1970 2019
13 expert.csv welf_vparty V-Party economic left-right pro_welfare pro_market pro_market no 3066 3066 585 65 1970 2019
14 lr_data.csv lr_ches CHES general left-right NA NA source-coded left-right no 1320 1320 390 44 1999 2024
15 lr_data.csv lr_morgan Morgan general left-right NA NA source-coded left-right no 471 471 72 11 1945 1973
16 lr_data.csv lr_poppa POPPA general left-right NA NA source-coded left-right no 416 416 225 31 2018 2023
17 text_data.csv conservative_morality_manifesto Manifesto Project cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 4501 4501 713 65 1920 2025
18 text_data.csv internationalism_manifesto Manifesto Project cultural cosmopolitan--traditionalist traditional cosmopolitan cosmopolitan yes 4501 4501 713 65 1920 2025
19 text_data.csv multiculturalism_manifesto Manifesto Project cultural cosmopolitan--traditionalist traditional cosmopolitan cosmopolitan yes 4501 4501 713 65 1920 2025
20 text_data.csv national_identity_manifesto Manifesto Project cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 4501 4501 713 65 1920 2025
21 text_data.csv economic_intervention_manifesto Manifesto Project economic left-right pro_market pro_welfare pro_welfare yes 4501 4501 713 65 1920 2025
22 text_data.csv economic_liberalization_manifesto Manifesto Project economic left-right pro_welfare pro_market pro_market no 4501 4501 713 65 1920 2025
23 text_data.csv market_regulation_manifesto Manifesto Project economic left-right pro_welfare pro_market pro_market no 4501 4501 713 65 1920 2025
24 text_data.csv social_services_manifesto Manifesto Project economic left-right pro_market pro_welfare pro_welfare yes 4501 4501 713 65 1920 2025
25 text_data.csv cultlib_poldem PolDem cultural cosmopolitan--traditionalist traditional cosmopolitan cosmopolitan yes 299 299 78 15 1972 2017
26 text_data.csv defense_poldem PolDem cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 243 243 67 15 1972 2017
27 text_data.csv euro_poldem PolDem cultural cosmopolitan--traditionalist traditional cosmopolitan cosmopolitan yes 93 93 44 13 1978 2017
28 text_data.csv europe_poldem PolDem cultural cosmopolitan--traditionalist traditional cosmopolitan cosmopolitan yes 217 217 66 15 1972 2017
29 text_data.csv immig_poldem PolDem cultural cosmopolitan--traditionalist traditional cosmopolitan cosmopolitan yes 236 236 67 14 1972 2017
30 text_data.csv nationalism_poldem PolDem cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 131 131 60 15 1974 2017
31 text_data.csv security_poldem PolDem cultural cosmopolitan--traditionalist cosmopolitan traditional traditional no 265 265 78 15 1972 2017
32 text_data.csv ecolib_poldem PolDem economic left-right pro_welfare pro_market pro_market no 361 361 85 15 1972 2017
33 text_data.csv welfare_poldem PolDem economic left-right pro_market pro_welfare pro_welfare yes 349 349 83 15 1972 2017
@@ -0,0 +1,9 @@
dimension,parameters,mean_rhat,max_rhat,min_ess_bulk,mean_ess_bulk
cultural cosmopolitan--traditionalist,17585,1.0006424726753271,1.0063876592524996,810.5088684922246,6889.24553993031
economic left-right,17585,1.0003816350246089,1.0041031832256586,670.7084347577414,5573.455663381927
lr_country_offset,65,1.0007983450724103,1.0042741902159762,1078.1975001602086,5434.867801353405
lr_decade_offset,8,1.0003240812347383,1.0011721739400503,2069.6763143867825,3372.015629437053
lr_sigma,3,1.0010806271267045,1.001817706171186,1956.6110527635497,2582.401306906664
lr_source_offset,3,1.0001329930739762,1.0002839044194227,3477.5194916377077,3814.373810300655
lr_weight,3,1.0023083674707072,1.0023587853631075,1540.8061256188755,1560.2173784478186
mean_sigma,6,1.00453801817315,1.0092946499280384,534.8105812273362,918.5017536375844
1 dimension parameters mean_rhat max_rhat min_ess_bulk mean_ess_bulk
2 cultural cosmopolitan--traditionalist 17585 1.0006424726753271 1.0063876592524996 810.5088684922246 6889.24553993031
3 economic left-right 17585 1.0003816350246089 1.0041031832256586 670.7084347577414 5573.455663381927
4 lr_country_offset 65 1.0007983450724103 1.0042741902159762 1078.1975001602086 5434.867801353405
5 lr_decade_offset 8 1.0003240812347383 1.0011721739400503 2069.6763143867825 3372.015629437053
6 lr_sigma 3 1.0010806271267045 1.001817706171186 1956.6110527635497 2582.401306906664
7 lr_source_offset 3 1.0001329930739762 1.0002839044194227 3477.5194916377077 3814.373810300655
8 lr_weight 3 1.0023083674707072 1.0023587853631075 1540.8061256188755 1560.2173784478186
9 mean_sigma 6 1.00453801817315 1.0092946499280384 534.8105812273362 918.5017536375844
@@ -0,0 +1,7 @@
metric,count,percentage
R-hat < 1.01,35258,100
R-hat 1.01-1.05,0,0
R-hat > 1.05,0,0
ESS > 1000,34853,98.85
ESS 400-1000,405,1.15
ESS < 400,0,0
1 metric count percentage
2 R-hat < 1.01 35258 100
3 R-hat 1.01-1.05 0 0
4 R-hat > 1.05 0 0
5 ESS > 1000 34853 98.85
6 ESS 400-1000 405 1.15
7 ESS < 400 0 0
@@ -0,0 +1,11 @@
family,parties,party_years,countries,min_year,max_year
soc,86,2892,37,1944,2025
con,82,2425,34,1944,2024
lib,81,2063,35,1944,2025
chr,41,1596,24,1945,2025
com,49,1241,27,1944,2025
right,50,975,26,1946,2025
eco,30,723,24,1960,2024
agr,12,592,10,1944,2024
spec,25,520,14,1949,2024
other,2,40,2,1992,2024
1 family parties party_years countries min_year max_year
2 soc 86 2892 37 1944 2025
3 con 82 2425 34 1944 2024
4 lib 81 2063 35 1944 2025
5 chr 41 1596 24 1945 2025
6 com 49 1241 27 1944 2025
7 right 50 975 26 1946 2025
8 eco 30 723 24 1960 2024
9 agr 12 592 10 1944 2024
10 spec 25 520 14 1949 2024
11 other 2 40 2 1992 2024
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,2 @@
rows,parties,countries,min_year,max_year,mean_economic_se,median_economic_se,mean_cultural_se,median_cultural_se
17585,708,65,1944,2025,0.06480914153714339,0.06297344495207625,0.05830289918910167,0.05511815999995913
1 rows parties countries min_year max_year mean_economic_se median_economic_se mean_cultural_se median_cultural_se
2 17585 708 65 1944 2025 0.06480914153714339 0.06297344495207625 0.05830289918910167 0.05511815999995913
@@ -0,0 +1,3 @@
dimension,r_pearson,r_spearman,ci_lower,ci_upper,mae,rmse,n,diagnostic
economic left-right,0.9040661964265828,0.9011355686083358,0.8986678208050985,0.9091907336941026,0.07569891833571618,0.09876331318961644,4637,convergent validity
cultural cosmopolitan--traditionalist,0.9598645558100498,0.96729961069876,0.9555654780202564,0.96375540322017,0.07890038540202159,0.0987420458613568,1425,convergent validity
1 dimension r_pearson r_spearman ci_lower ci_upper mae rmse n diagnostic
2 economic left-right 0.9040661964265828 0.9011355686083358 0.8986678208050985 0.9091907336941026 0.07569891833571618 0.09876331318961644 4637 convergent validity
3 cultural cosmopolitan--traditionalist 0.9598645558100498 0.96729961069876 0.9555654780202564 0.96375540322017 0.07890038540202159 0.0987420458613568 1425 convergent validity
@@ -0,0 +1,5 @@
model_dim,expert_dim,r_pearson,r_spearman,n,type,diagnostic
economic left-right,economic,0.9040661964265828,0.9011355686083358,4637,convergent,discriminant validity
cultural cosmopolitan--traditionalist,economic,0.42228411618900114,0.4352014815908644,4637,discriminant,discriminant validity
cultural cosmopolitan--traditionalist,cultural cosmopolitan--traditionalist,0.9598645558100498,0.96729961069876,1425,convergent,discriminant validity
economic left-right,cultural cosmopolitan--traditionalist,0.39267427502305646,0.3979530504947876,1425,discriminant,discriminant validity
1 model_dim expert_dim r_pearson r_spearman n type diagnostic
2 economic left-right economic 0.9040661964265828 0.9011355686083358 4637 convergent discriminant validity
3 cultural cosmopolitan--traditionalist economic 0.42228411618900114 0.4352014815908644 4637 discriminant discriminant validity
4 cultural cosmopolitan--traditionalist cultural cosmopolitan--traditionalist 0.9598645558100498 0.96729961069876 1425 convergent discriminant validity
5 economic left-right cultural cosmopolitan--traditionalist 0.39267427502305646 0.3979530504947876 1425 discriminant discriminant validity
@@ -0,0 +1,3 @@
dimension,cic,cic_pct,ci_lower,ci_upper,n,covered,diagnostic
economic left-right,0.8987256874580818,89.9,0.8916705123992721,0.9053701433508112,7455,6700,posterior predictive coverage
cultural cosmopolitan--traditionalist,0.8477379496750113,84.8,0.8420030513784423,0.8533009527490912,15539,13173,posterior predictive coverage
1 dimension cic cic_pct ci_lower ci_upper n covered diagnostic
2 economic left-right 0.8987256874580818 89.9 0.8916705123992721 0.9053701433508112 7455 6700 posterior predictive coverage
3 cultural cosmopolitan--traditionalist 0.8477379496750113 84.8 0.8420030513784423 0.8533009527490912 15539 13173 posterior predictive coverage
+10
View File
@@ -0,0 +1,10 @@
source_file,item,source,dimension,type_low,type_high,higher_values_indicate,reversed_for_reporting,observations,party_years,parties,countries,min_year,max_year
text_data.csv,internationalism_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,4501,4501,713,65,1920,2025
text_data.csv,multiculturalism_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,4501,4501,713,65,1920,2025
text_data.csv,economic_intervention_manifesto,Manifesto Project,economic left-right,pro_market,pro_welfare,pro_welfare,yes,4501,4501,713,65,1920,2025
text_data.csv,social_services_manifesto,Manifesto Project,economic left-right,pro_market,pro_welfare,pro_welfare,yes,4501,4501,713,65,1920,2025
text_data.csv,cultlib_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,299,299,78,15,1972,2017
text_data.csv,euro_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,93,93,44,13,1978,2017
text_data.csv,europe_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,217,217,66,15,1972,2017
text_data.csv,immig_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,236,236,67,14,1972,2017
text_data.csv,welfare_poldem,PolDem,economic left-right,pro_market,pro_welfare,pro_welfare,yes,349,349,83,15,1972,2017
1 source_file item source dimension type_low type_high higher_values_indicate reversed_for_reporting observations party_years parties countries min_year max_year
2 text_data.csv internationalism_manifesto Manifesto Project cultural cosmopolitan--traditionalist traditional cosmopolitan cosmopolitan yes 4501 4501 713 65 1920 2025
3 text_data.csv multiculturalism_manifesto Manifesto Project cultural cosmopolitan--traditionalist traditional cosmopolitan cosmopolitan yes 4501 4501 713 65 1920 2025
4 text_data.csv economic_intervention_manifesto Manifesto Project economic left-right pro_market pro_welfare pro_welfare yes 4501 4501 713 65 1920 2025
5 text_data.csv social_services_manifesto Manifesto Project economic left-right pro_market pro_welfare pro_welfare yes 4501 4501 713 65 1920 2025
6 text_data.csv cultlib_poldem PolDem cultural cosmopolitan--traditionalist traditional cosmopolitan cosmopolitan yes 299 299 78 15 1972 2017
7 text_data.csv euro_poldem PolDem cultural cosmopolitan--traditionalist traditional cosmopolitan cosmopolitan yes 93 93 44 13 1978 2017
8 text_data.csv europe_poldem PolDem cultural cosmopolitan--traditionalist traditional cosmopolitan cosmopolitan yes 217 217 66 15 1972 2017
9 text_data.csv immig_poldem PolDem cultural cosmopolitan--traditionalist traditional cosmopolitan cosmopolitan yes 236 236 67 14 1972 2017
10 text_data.csv welfare_poldem PolDem economic left-right pro_market pro_welfare pro_welfare yes 349 349 83 15 1972 2017
@@ -0,0 +1,7 @@
specification,ablated_source,dimension,matched_n,correlation_with_production,mean_abs_difference,median_abs_difference,p95_abs_difference,mean_interval_width_production,mean_interval_width_ablation
Source ablation,V Party,economic left-right,4248,0.954,0.05,0.033,0.159,NA,NA
Source ablation,V Party,cultural cosmopolitan--traditionalist,4248,0.898,0.08,0.062,0.228,0.202,0.323
Gap threshold 5 years,NA,economic left-right,4244,0.999,0.003,0.002,0.007,NA,NA
Gap threshold 5 years,NA,cultural cosmopolitan--traditionalist,4244,0.999,0.002,0.001,0.006,NA,NA
Gap threshold 10 years,NA,economic left-right,4265,0.999,0.006,0.006,0.011,NA,NA
Gap threshold 10 years,NA,cultural cosmopolitan--traditionalist,4265,1,0.003,0.002,0.006,NA,NA
1 specification ablated_source dimension matched_n correlation_with_production mean_abs_difference median_abs_difference p95_abs_difference mean_interval_width_production mean_interval_width_ablation
2 Source ablation V Party economic left-right 4248 0.954 0.05 0.033 0.159 NA NA
3 Source ablation V Party cultural cosmopolitan--traditionalist 4248 0.898 0.08 0.062 0.228 0.202 0.323
4 Gap threshold 5 years NA economic left-right 4244 0.999 0.003 0.002 0.007 NA NA
5 Gap threshold 5 years NA cultural cosmopolitan--traditionalist 4244 0.999 0.002 0.001 0.006 NA NA
6 Gap threshold 10 years NA economic left-right 4265 0.999 0.006 0.006 0.011 NA NA
7 Gap threshold 10 years NA cultural cosmopolitan--traditionalist 4265 1 0.003 0.002 0.006 NA NA
@@ -0,0 +1,7 @@
dimension,source_composition_class,reference_class,n,adjusted_difference
economic left-right,text_only_direct_or_nearby,both_direct_or_nearby,4916,0.014
economic left-right,expert_only_direct_or_nearby,both_direct_or_nearby,4916,-0.015
economic left-right,temporal_propagation,both_direct_or_nearby,4916,-0.046
cultural cosmopolitan--traditionalist,text_only_direct_or_nearby,both_direct_or_nearby,4916,0.017
cultural cosmopolitan--traditionalist,expert_only_direct_or_nearby,both_direct_or_nearby,4916,0.038
cultural cosmopolitan--traditionalist,temporal_propagation,both_direct_or_nearby,4916,-0.042
1 dimension source_composition_class reference_class n adjusted_difference
2 economic left-right text_only_direct_or_nearby both_direct_or_nearby 4916 0.014
3 economic left-right expert_only_direct_or_nearby both_direct_or_nearby 4916 -0.015
4 economic left-right temporal_propagation both_direct_or_nearby 4916 -0.046
5 cultural cosmopolitan--traditionalist text_only_direct_or_nearby both_direct_or_nearby 4916 0.017
6 cultural cosmopolitan--traditionalist expert_only_direct_or_nearby both_direct_or_nearby 4916 0.038
7 cultural cosmopolitan--traditionalist temporal_propagation both_direct_or_nearby 4916 -0.042
+10
View File
@@ -0,0 +1,10 @@
source_file,source,items,observations,party_years,parties,countries,min_year,max_year
expert.csv,CHES,2,2639,1320,390,44,1999,2024
expert.csv,GPS,2,538,271,271,62,2019,2019
expert.csv,POPPA,1,413,413,225,31,2018,2023
expert.csv,V-Party,7,21488,3076,589,65,1970,2019
lr_data.csv,CHES,1,1320,1320,390,44,1999,2024
lr_data.csv,Morgan,1,471,471,72,11,1945,1973
lr_data.csv,POPPA,1,416,416,225,31,2018,2023
text_data.csv,Manifesto Project,8,36008,4501,713,65,1920,2025
text_data.csv,PolDem,9,2194,406,93,15,1972,2017
1 source_file source items observations party_years parties countries min_year max_year
2 expert.csv CHES 2 2639 1320 390 44 1999 2024
3 expert.csv GPS 2 538 271 271 62 2019 2019
4 expert.csv POPPA 1 413 413 225 31 2018 2023
5 expert.csv V-Party 7 21488 3076 589 65 1970 2019
6 lr_data.csv CHES 1 1320 1320 390 44 1999 2024
7 lr_data.csv Morgan 1 471 471 72 11 1945 1973
8 lr_data.csv POPPA 1 416 416 225 31 2018 2023
9 text_data.csv Manifesto Project 8 36008 4501 713 65 1920 2025
10 text_data.csv PolDem 9 2194 406 93 15 1972 2017
+66
View File
@@ -0,0 +1,66 @@
"release","iso2","country","first_year","last_year","parties","election_year_rows","both_text_expert","text_only","expert_only","temporal_propagation"
"v0","AL","Albania",1991,2021,9,51,19,23,9,0
"v0","AM","Armenia",1995,2021,8,26,25,1,0,0
"v0","AR","Argentina",1983,2019,9,51,27,6,9,9
"v0","AT","Austria",1949,2019,7,84,64,20,0,0
"v0","AU","Australia",1946,2022,9,129,70,53,6,0
"v0","AZ","Azerbaijan",1995,2015,3,9,5,1,3,0
"v0","BA","Bosnia and Herzegovina",1990,2022,9,59,43,16,0,0
"v0","BE","Belgium",1946,2019,20,194,114,77,3,0
"v0","BG","Bulgaria",1990,2017,12,43,37,5,0,1
"v0","BO","Bolivia",1979,2014,2,13,2,1,10,0
"v0","BR","Brazil",1982,2022,8,69,3,57,6,3
"v0","BY","Belarus",1995,2008,1,4,1,0,3,0
"v0","CA","Canada",1945,2021,8,104,64,39,1,0
"v0","CH","Switzerland",1947,2019,16,155,69,86,0,0
"v0","CL","Chile",1989,2021,9,65,0,63,0,2
"v0","CO","Colombia",1970,2018,9,45,12,9,24,0
"v0","CR","Costa Rica",1970,2018,7,35,29,0,6,0
"v0","CY","Cyprus",1970,2016,7,44,22,4,18,0
"v0","CZ","Czechia",1990,2017,12,51,50,1,0,0
"v0","DE","Germany",1949,2025,11,115,56,59,0,0
"v0","DK","Denmark",1945,2019,17,236,131,105,0,0
"v0","DO","Dominican Republic",1978,2016,3,38,29,0,9,0
"v0","EC","Ecuador",1979,2017,9,66,36,4,20,6
"v0","EE","Estonia",1992,2019,12,50,41,6,3,0
"v0","ES","Spain",1977,2023,23,154,108,45,0,1
"v0","FI","Finland",1945,2019,11,158,88,70,0,0
"v0","FR","France",1946,2022,16,121,62,55,4,0
"v0","GB","United Kingdom",1945,2024,12,106,71,35,0,0
"v0","GE","Georgia",1992,2020,10,27,24,3,0,0
"v0","GR","Greece",1974,2023,13,76,69,4,2,1
"v0","HR","Croatia",1990,2020,13,60,41,15,3,1
"v0","HU","Hungary",1990,2022,12,50,42,7,1,0
"v0","IE","Ireland",1948,2020,12,100,63,37,0,0
"v0","IL","Israel",1949,2022,33,218,91,102,12,13
"v0","IS","Iceland",1946,2021,14,116,84,32,0,0
"v0","IT","Italy",1946,2018,27,132,55,69,0,8
"v0","JP","Japan",1960,2021,12,120,80,40,0,0
"v0","KR","South Korea",1992,2020,10,25,20,5,0,0
"v0","LK","Sri Lanka",1947,2015,4,32,8,17,7,0
"v0","LT","Lithuania",1992,2020,13,50,43,2,5,0
"v0","LU","Luxembourg",1945,2013,7,74,44,30,0,0
"v0","LV","Latvia",1993,2022,16,55,41,7,1,6
"v0","MD","Moldova",1994,2019,6,22,22,0,0,0
"v0","ME","Montenegro",1990,2023,13,50,32,18,0,0
"v0","MK","North Macedonia",1990,2016,11,57,33,23,1,0
"v0","MT","Malta",1971,2022,2,24,14,0,10,0
"v0","MX","Mexico",1946,2018,10,86,31,31,1,23
"v0","NL","Netherlands",1946,2021,23,188,123,65,0,0
"v0","NO","Norway",1945,2017,10,129,71,58,0,0
"v0","NZ","New Zealand",1946,2020,10,108,64,42,1,1
"v0","PA","Panama",1980,2019,6,26,4,8,14,0
"v0","PE","Peru",1978,2016,5,18,7,0,11,0
"v0","PL","Poland",1972,2019,12,53,37,10,6,0
"v0","PT","Portugal",1975,2022,15,121,73,47,1,0
"v0","RO","Romania",1990,2016,11,35,19,5,1,10
"v0","RS","Serbia",1990,2023,18,80,41,39,0,0
"v0","RU","Russia",1993,2011,10,31,28,3,0,0
"v0","SE","Sweden",1944,2022,8,144,103,41,0,0
"v0","SI","Slovenia",1990,2018,15,70,60,5,5,0
"v0","SK","Slovakia",1990,2016,14,54,47,7,0,0
"v0","TR","Türkiye",1950,2018,12,65,44,17,4,0
"v0","UA","Ukraine",1994,2019,11,35,31,3,1,0
"v0","US","United States",1944,2024,2,78,52,26,0,0
"v0","UY","Uruguay",1984,2014,3,16,2,0,14,0
"v0","ZA","South Africa",1989,2019,6,24,18,4,2,0
1 release iso2 country first_year last_year parties election_year_rows both_text_expert text_only expert_only temporal_propagation
2 v0 AL Albania 1991 2021 9 51 19 23 9 0
3 v0 AM Armenia 1995 2021 8 26 25 1 0 0
4 v0 AR Argentina 1983 2019 9 51 27 6 9 9
5 v0 AT Austria 1949 2019 7 84 64 20 0 0
6 v0 AU Australia 1946 2022 9 129 70 53 6 0
7 v0 AZ Azerbaijan 1995 2015 3 9 5 1 3 0
8 v0 BA Bosnia and Herzegovina 1990 2022 9 59 43 16 0 0
9 v0 BE Belgium 1946 2019 20 194 114 77 3 0
10 v0 BG Bulgaria 1990 2017 12 43 37 5 0 1
11 v0 BO Bolivia 1979 2014 2 13 2 1 10 0
12 v0 BR Brazil 1982 2022 8 69 3 57 6 3
13 v0 BY Belarus 1995 2008 1 4 1 0 3 0
14 v0 CA Canada 1945 2021 8 104 64 39 1 0
15 v0 CH Switzerland 1947 2019 16 155 69 86 0 0
16 v0 CL Chile 1989 2021 9 65 0 63 0 2
17 v0 CO Colombia 1970 2018 9 45 12 9 24 0
18 v0 CR Costa Rica 1970 2018 7 35 29 0 6 0
19 v0 CY Cyprus 1970 2016 7 44 22 4 18 0
20 v0 CZ Czechia 1990 2017 12 51 50 1 0 0
21 v0 DE Germany 1949 2025 11 115 56 59 0 0
22 v0 DK Denmark 1945 2019 17 236 131 105 0 0
23 v0 DO Dominican Republic 1978 2016 3 38 29 0 9 0
24 v0 EC Ecuador 1979 2017 9 66 36 4 20 6
25 v0 EE Estonia 1992 2019 12 50 41 6 3 0
26 v0 ES Spain 1977 2023 23 154 108 45 0 1
27 v0 FI Finland 1945 2019 11 158 88 70 0 0
28 v0 FR France 1946 2022 16 121 62 55 4 0
29 v0 GB United Kingdom 1945 2024 12 106 71 35 0 0
30 v0 GE Georgia 1992 2020 10 27 24 3 0 0
31 v0 GR Greece 1974 2023 13 76 69 4 2 1
32 v0 HR Croatia 1990 2020 13 60 41 15 3 1
33 v0 HU Hungary 1990 2022 12 50 42 7 1 0
34 v0 IE Ireland 1948 2020 12 100 63 37 0 0
35 v0 IL Israel 1949 2022 33 218 91 102 12 13
36 v0 IS Iceland 1946 2021 14 116 84 32 0 0
37 v0 IT Italy 1946 2018 27 132 55 69 0 8
38 v0 JP Japan 1960 2021 12 120 80 40 0 0
39 v0 KR South Korea 1992 2020 10 25 20 5 0 0
40 v0 LK Sri Lanka 1947 2015 4 32 8 17 7 0
41 v0 LT Lithuania 1992 2020 13 50 43 2 5 0
42 v0 LU Luxembourg 1945 2013 7 74 44 30 0 0
43 v0 LV Latvia 1993 2022 16 55 41 7 1 6
44 v0 MD Moldova 1994 2019 6 22 22 0 0 0
45 v0 ME Montenegro 1990 2023 13 50 32 18 0 0
46 v0 MK North Macedonia 1990 2016 11 57 33 23 1 0
47 v0 MT Malta 1971 2022 2 24 14 0 10 0
48 v0 MX Mexico 1946 2018 10 86 31 31 1 23
49 v0 NL Netherlands 1946 2021 23 188 123 65 0 0
50 v0 NO Norway 1945 2017 10 129 71 58 0 0
51 v0 NZ New Zealand 1946 2020 10 108 64 42 1 1
52 v0 PA Panama 1980 2019 6 26 4 8 14 0
53 v0 PE Peru 1978 2016 5 18 7 0 11 0
54 v0 PL Poland 1972 2019 12 53 37 10 6 0
55 v0 PT Portugal 1975 2022 15 121 73 47 1 0
56 v0 RO Romania 1990 2016 11 35 19 5 1 10
57 v0 RS Serbia 1990 2023 18 80 41 39 0 0
58 v0 RU Russia 1993 2011 10 31 28 3 0 0
59 v0 SE Sweden 1944 2022 8 144 103 41 0 0
60 v0 SI Slovenia 1990 2018 15 70 60 5 5 0
61 v0 SK Slovakia 1990 2016 14 54 47 7 0 0
62 v0 TR Türkiye 1950 2018 12 65 44 17 4 0
63 v0 UA Ukraine 1994 2019 11 35 31 3 1 0
64 v0 US United States 1944 2024 2 78 52 26 0 0
65 v0 UY Uruguay 1984 2014 3 16 2 0 14 0
66 v0 ZA South Africa 1989 2019 6 24 18 4 2 0
+28
View File
@@ -0,0 +1,28 @@
file_name,variable_name,label,description,type,allowed_values,range_min,range_max,missing_value_code,unit,scale_direction,constructed_from,construction_rule,notes
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),party_id,PartyFacts party identifier,Identifier for the individual party,integer,,,,,identifier,,PartyFacts crosswalk,Assigned during party harmonization,
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),party_name_english,English party name,Party name from the harmonized output,string,,,,,name,,Party metadata,,
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),party_name_short,Short party name,Short party label from the harmonized output,string,,,,,name,,Party metadata,,May be missing
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),country,Country code,ISO2 or historical country-code identifier,string,,,,,identifier,,Source metadata and PartyFacts,,
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),year,Calendar year,Calendar year of the election-year estimate,integer,,1944,2025,,year,,Election and model-year metadata,,
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),segment_num,Party segment number,Segment number within party after splitting at long evidence gaps,integer,,1,,,segment,,Party history segmentation,Main segment is coded 1,
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),union_party_id,Union or alliance PartyFacts identifier,Identifier of parent union or alliance where applicable,integer,,,,,identifier,,Alliance mapping,,Missing if party is not represented through a union or alliance
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),in_union,Union membership indicator,Indicator that the row is associated with a union or alliance,boolean,0;1,0,1,,indicator,,Alliance mapping,,
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),pervote,Vote share,Vote share at the election year,numeric,,0,100,,percent,,Election metadata,,
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),economic_lr,Economic left-right posterior mean,Posterior mean of economic left-right position,numeric,,0,1,,unit interval,0=left; 1=right,Posterior draws,Mean after inverse-logit transformation,
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),galtan,Cultural cosmopolitan--traditionalist posterior mean,Posterior mean of cultural cosmopolitan--traditionalist position,numeric,,0,1,,unit interval,0=cosmopolitan; 1=traditionalist,Posterior draws,Mean after inverse-logit transformation,
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),economic_lr_se,Economic posterior standard error,Posterior standard deviation for economic_lr,numeric,,0,,,unit interval,,Posterior draws,Standard deviation over posterior draws,
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),galtan_se,Cultural posterior standard error,Posterior standard deviation for the cultural estimate,numeric,,0,,,unit interval,,Posterior draws,Standard deviation over posterior draws,
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),economic_lr_q025,Economic lower posterior interval,2.5 percent posterior quantile for economic_lr,numeric,,0,1,,unit interval,0=left; 1=right,Posterior draws,Quantile over posterior draws,
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),economic_lr_q975,Economic upper posterior interval,97.5 percent posterior quantile for economic_lr,numeric,,0,1,,unit interval,0=left; 1=right,Posterior draws,Quantile over posterior draws,
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),galtan_q025,Cultural lower posterior interval,2.5 percent posterior quantile for the cultural estimate,numeric,,0,1,,unit interval,0=cosmopolitan; 1=traditionalist,Posterior draws,Quantile over posterior draws,
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),galtan_q975,Cultural upper posterior interval,97.5 percent posterior quantile for the cultural estimate,numeric,,0,1,,unit interval,0=cosmopolitan; 1=traditionalist,Posterior draws,Quantile over posterior draws,
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),election_id,Election identifier,Election identifier where available,string,,,,,identifier,,Election metadata,,Missing where no election identifier is available
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),election_date,Election date,Election date where available,date,,,,,date,,Election metadata,,Missing in current processed election metadata
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),has_text,Text support indicator,Indicator for direct or nearby text evidence,boolean,0;1,0,1,,indicator,,Source-support construction,Nearby threshold documented in source_support_dictionary.csv,
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),has_expert,Expert support indicator,Indicator for direct or nearby expert evidence,boolean,0;1,0,1,,indicator,,Source-support construction,Nearby threshold documented in source_support_dictionary.csv,
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),n_text_sources,Number of text sources,Number of distinct text source families contributing direct or nearby evidence,integer,,0,,,count,,Source-support construction,,
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),n_expert_sources,Number of expert sources,Number of distinct expert source families contributing direct or nearby evidence,integer,,0,,,count,,Source-support construction,,
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),nearest_text_distance,Nearest text distance,Absolute distance in years to nearest text observation used to inform trajectory,numeric,,0,,,years,,Source-support construction,,Missing if no text observation exists for the party-country key
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),nearest_expert_distance,Nearest expert distance,Absolute distance in years to nearest expert observation used to inform trajectory,numeric,,0,,,years,,Source-support construction,,Missing if no expert observation exists for the party-country key
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),source_support_class,Source-support class,Summary class of text/expert support,string,both_direct_or_nearby; text_only_direct_or_nearby; expert_only_direct_or_nearby; temporal_propagation,,,,categorical,,Source-support construction,,
party_2d_annual_model_output_vN.csv (inside party_2d_annual_model_output_vN.zip),all fields,Annual model output fields,Same model-output fields as the production annual party-position file plus no source-support augmentation,mixed,,,,,,,Posterior processing,,Secondary model output
1 file_name variable_name label description type allowed_values range_min range_max missing_value_code unit scale_direction constructed_from construction_rule notes
2 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) party_id PartyFacts party identifier Identifier for the individual party integer identifier PartyFacts crosswalk Assigned during party harmonization
3 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) party_name_english English party name Party name from the harmonized output string name Party metadata
4 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) party_name_short Short party name Short party label from the harmonized output string name Party metadata May be missing
5 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) country Country code ISO2 or historical country-code identifier string identifier Source metadata and PartyFacts
6 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) year Calendar year Calendar year of the election-year estimate integer 1944 2025 year Election and model-year metadata
7 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) segment_num Party segment number Segment number within party after splitting at long evidence gaps integer 1 segment Party history segmentation Main segment is coded 1
8 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) union_party_id Union or alliance PartyFacts identifier Identifier of parent union or alliance where applicable integer identifier Alliance mapping Missing if party is not represented through a union or alliance
9 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) in_union Union membership indicator Indicator that the row is associated with a union or alliance boolean 0;1 0 1 indicator Alliance mapping
10 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) pervote Vote share Vote share at the election year numeric 0 100 percent Election metadata
11 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) economic_lr Economic left-right posterior mean Posterior mean of economic left-right position numeric 0 1 unit interval 0=left; 1=right Posterior draws Mean after inverse-logit transformation
12 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) galtan Cultural cosmopolitan--traditionalist posterior mean Posterior mean of cultural cosmopolitan--traditionalist position numeric 0 1 unit interval 0=cosmopolitan; 1=traditionalist Posterior draws Mean after inverse-logit transformation
13 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) economic_lr_se Economic posterior standard error Posterior standard deviation for economic_lr numeric 0 unit interval Posterior draws Standard deviation over posterior draws
14 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) galtan_se Cultural posterior standard error Posterior standard deviation for the cultural estimate numeric 0 unit interval Posterior draws Standard deviation over posterior draws
15 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) economic_lr_q025 Economic lower posterior interval 2.5 percent posterior quantile for economic_lr numeric 0 1 unit interval 0=left; 1=right Posterior draws Quantile over posterior draws
16 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) economic_lr_q975 Economic upper posterior interval 97.5 percent posterior quantile for economic_lr numeric 0 1 unit interval 0=left; 1=right Posterior draws Quantile over posterior draws
17 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) galtan_q025 Cultural lower posterior interval 2.5 percent posterior quantile for the cultural estimate numeric 0 1 unit interval 0=cosmopolitan; 1=traditionalist Posterior draws Quantile over posterior draws
18 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) galtan_q975 Cultural upper posterior interval 97.5 percent posterior quantile for the cultural estimate numeric 0 1 unit interval 0=cosmopolitan; 1=traditionalist Posterior draws Quantile over posterior draws
19 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) election_id Election identifier Election identifier where available string identifier Election metadata Missing where no election identifier is available
20 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) election_date Election date Election date where available date date Election metadata Missing in current processed election metadata
21 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) has_text Text support indicator Indicator for direct or nearby text evidence boolean 0;1 0 1 indicator Source-support construction Nearby threshold documented in source_support_dictionary.csv
22 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) has_expert Expert support indicator Indicator for direct or nearby expert evidence boolean 0;1 0 1 indicator Source-support construction Nearby threshold documented in source_support_dictionary.csv
23 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) n_text_sources Number of text sources Number of distinct text source families contributing direct or nearby evidence integer 0 count Source-support construction
24 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) n_expert_sources Number of expert sources Number of distinct expert source families contributing direct or nearby evidence integer 0 count Source-support construction
25 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) nearest_text_distance Nearest text distance Absolute distance in years to nearest text observation used to inform trajectory numeric 0 years Source-support construction Missing if no text observation exists for the party-country key
26 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) nearest_expert_distance Nearest expert distance Absolute distance in years to nearest expert observation used to inform trajectory numeric 0 years Source-support construction Missing if no expert observation exists for the party-country key
27 party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip) source_support_class Source-support class Summary class of text/expert support string both_direct_or_nearby; text_only_direct_or_nearby; expert_only_direct_or_nearby; temporal_propagation categorical Source-support construction
28 party_2d_annual_model_output_vN.csv (inside party_2d_annual_model_output_vN.zip) all fields Annual model output fields Same model-output fields as the production annual party-position file plus no source-support augmentation mixed Posterior processing Secondary model output
+9
View File
@@ -0,0 +1,9 @@
field,value,definition,notes
direct_support,observation in same calendar year,Observation in the same calendar year as the election-year estimate,
nearby_support,observation within threshold,Observation within the stated distance threshold of the election-year estimate,
nearby_support_threshold_years,2,Numerical threshold in years used to classify nearby support,
temporal_propagation,no nearby observation,No text or expert observation within the nearby threshold,
source_support_class,both_direct_or_nearby,Both text and expert evidence are direct or nearby,
source_support_class,text_only_direct_or_nearby,Text evidence is direct or nearby and expert evidence is not,
source_support_class,expert_only_direct_or_nearby,Expert evidence is direct or nearby and text evidence is not,
source_support_class,temporal_propagation,Neither text nor expert evidence is direct or nearby,
1 field value definition notes
2 direct_support observation in same calendar year Observation in the same calendar year as the election-year estimate
3 nearby_support observation within threshold Observation within the stated distance threshold of the election-year estimate
4 nearby_support_threshold_years 2 Numerical threshold in years used to classify nearby support
5 temporal_propagation no nearby observation No text or expert observation within the nearby threshold
6 source_support_class both_direct_or_nearby Both text and expert evidence are direct or nearby
7 source_support_class text_only_direct_or_nearby Text evidence is direct or nearby and expert evidence is not
8 source_support_class expert_only_direct_or_nearby Expert evidence is direct or nearby and text evidence is not
9 source_support_class temporal_propagation Neither text nor expert evidence is direct or nearby
+595
View File
@@ -0,0 +1,595 @@
// =============================================================================
// 2D BIPOLAR LATENT TRAIT MODEL V6 - Hierarchical L-R Weights
// =============================================================================
//
// V6 CHANGE: Hierarchical L-R weights varying by country, source, and decade
// - Replaces global simplex[2] lr_weights with additive logit-scale model
// - logit_weight[n] = global + country_offset[c] + source_offset[k] + decade_offset[d]
// - All offsets use non-centered parameterization with estimated sigma hyperparameters
// - Data-poor contexts shrink toward global mean (~55 new scalar parameters)
//
// PRESERVED FROM V5:
// - Beta-Binomial expert likelihood with K-scaling
// - V-Party cultural expansion (5 native items + v2pawelf)
// - Mean-constituent model (individual party estimates for union members)
// - Segment-based indexing (S segments, R segment-years)
// - 3-level hierarchical variance (global -> country -> family)
// - Random walk dynamics within segments
// - CDU anchor for scale identification (CDU=1375, not CDU/CSU=211)
// - Non-centered parameterization for efficiency
// - Zero-inflation model for manifesto data
// - Country-item intercepts (coding convention differences across countries)
// - Binomial-logit likelihood for text data (unchanged)
//
// =============================================================================
data {
// Segment structure
int<lower=1> S; // Number of segments
int<lower=1> P; // Number of countries
int<lower=1> R; // Total unique segment-year combinations
int<lower=1> T_year; // Total number of years
array[S] int<lower=1> len_theta_ts; // Number of years per segment
// Country membership for each segment
array[S] int<lower=1, upper=P> segment_country;
// Segment family data
int<lower=1> F; // Number of party families
array[S] int<lower=1, upper=F> segment_family; // Family index for each segment
// =========================================================================
// Text data (manifesto + PolDem)
// =========================================================================
int<lower=1> N_man; // Number of text observations
int<lower=1> K_man; // Number of unique text items
array[N_man] int<lower=1, upper=K_man> kk_man; // Text item index
array[N_man] int<lower=1, upper=S> ss_man; // Segment index (first constituent for unions)
array[N_man] int<lower=1, upper=P> pp_man; // Country index
array[N_man] int<lower=0> positive; // Positive mentions
array[N_man] int<lower=0> sample; // Total sample size
array[N_man] int<lower=1, upper=T_year> year_for_man; // Year index
// Dimension and direction for text data
array[N_man] int<lower=1, upper=2> dim_idx_man; // 1=economic, 2=galtan
array[N_man] int<lower=-1, upper=1> direction_man; // +1=right/TAN, -1=left/GAL
// V4: Constituent structure for manifesto observations
int<lower=1> N_const_man_total; // Total entries in const_rr_man
array[N_man] int<lower=1> n_const_man; // Number of constituents per obs
array[N_man] int<lower=1> const_offset_man; // Offset into const_rr_man
array[N_const_man_total] int<lower=1, upper=R> const_rr_man; // Constituent rr indices
// Country-item-year data (used by zero-inflation model)
int<lower=1> N_ciy; // Number of unique country-item-year combinations
array[N_man] int<lower=1, upper=N_ciy> ciy_idx;
// =========================================================================
// Expert dimension-specific data (V5: integer observations + scale size)
// =========================================================================
int<lower=1> N_exp_dim; // Number of dimension-specific expert observations
int<lower=1> K_exp_dim; // Number of unique expert items
array[N_exp_dim] int<lower=1, upper=K_exp_dim> kk_exp_dim;
array[N_exp_dim] int<lower=1, upper=S> ss_exp_dim;
array[N_exp_dim] int<lower=1, upper=P> pp_exp_dim;
array[N_exp_dim] int<lower=0> val_dim_int; // V5: rounded sum = round(mean * K * n_scale)
array[N_exp_dim] int<lower=1> n_total_exp_dim; // V5: K * n_scale (total trials)
array[N_exp_dim] int<lower=1> n_experts_exp_dim; // V5: K (number of experts)
// Dimension index for expert data
array[N_exp_dim] int<lower=1, upper=2> dim_idx_exp; // 1=economic, 2=galtan
// V4: Constituent structure for expert dim observations
int<lower=1> N_const_exp_dim_total;
array[N_exp_dim] int<lower=1> n_const_exp_dim;
array[N_exp_dim] int<lower=1> const_offset_exp_dim;
array[N_const_exp_dim_total] int<lower=1, upper=R> const_rr_exp_dim;
// =========================================================================
// Expert general L-R data (V5: integer observations + scale size)
// =========================================================================
int<lower=1> N_exp_lr;
int<lower=1> K_exp_lr;
array[N_exp_lr] int<lower=1, upper=K_exp_lr> kk_exp_lr;
array[N_exp_lr] int<lower=1, upper=S> ss_exp_lr;
array[N_exp_lr] int<lower=1, upper=P> pp_exp_lr;
array[N_exp_lr] int<lower=0> val_lr_int; // V5: rounded sum = round(mean * K * n_scale)
array[N_exp_lr] int<lower=1> n_total_exp_lr; // V5: K * n_scale (total trials)
array[N_exp_lr] int<lower=1> n_experts_exp_lr; // V5: K (number of experts)
// V6: Decade indexing for hierarchical L-R weights
int<lower=1> D_lr; // Number of decades
array[N_exp_lr] int<lower=1, upper=D_lr> dd_exp_lr; // Decade index per LR obs
// V4: Constituent structure for expert L-R observations
int<lower=1> N_const_exp_lr_total;
array[N_exp_lr] int<lower=1> n_const_exp_lr;
array[N_exp_lr] int<lower=1> const_offset_exp_lr;
array[N_const_exp_lr_total] int<lower=1, upper=R> const_rr_exp_lr;
// Prior information
real mn_resp_log_man;
real mn_resp_log_exp_dim;
real mn_resp_log_exp_lr;
// Identification anchor
int<lower=1, upper=S> anchor_segment; // CDU segment (1375 with unions, 211 without)
// V3 compatibility: these are still passed but not used in V4+ likelihood
// (kept so older data dicts work without modification for backwards compat)
array[N_man] int<lower=1, upper=R> rr_man; // Segment-year index (unused in V4+ likelihood)
array[N_exp_dim] int<lower=1, upper=R> rr_exp_dim;
array[N_exp_lr] int<lower=1, upper=R> rr_exp_lr;
}
transformed data {
real eps = 1e-6;
real one_minus_eps = 1 - eps;
// Count zero and non-zero samples
int N_man_zero = 0;
int N_man_nonzero = 0;
for (n in 1:N_man) {
if (sample[n] == 0) {
N_man_zero += 1;
} else {
N_man_nonzero += 1;
}
}
// Create index arrays for zero/nonzero split
array[N_man_zero > 0 ? N_man_zero : 1] int idx_zero;
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int idx_nonzero;
{
int pos_zero = 1;
int pos_nonzero = 1;
for (n in 1:N_man) {
if (sample[n] == 0) {
idx_zero[pos_zero] = n;
pos_zero += 1;
} else {
idx_nonzero[pos_nonzero] = n;
pos_nonzero += 1;
}
}
}
// V4: Pre-compute constituent info for nonzero manifesto obs
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int kk_man_nonzero;
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int orig_idx_nonzero;
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int direction_nonzero;
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int pp_man_nonzero;
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int dim_idx_nonzero;
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int n_const_nonzero;
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int const_offset_nonzero;
{
for (i in 1:N_man_nonzero) {
int n = idx_nonzero[i];
orig_idx_nonzero[i] = n;
kk_man_nonzero[i] = kk_man[n];
direction_nonzero[i] = direction_man[n];
pp_man_nonzero[i] = pp_man[n];
dim_idx_nonzero[i] = dim_idx_man[n];
n_const_nonzero[i] = n_const_man[n];
const_offset_nonzero[i] = const_offset_man[n];
}
}
// Pre-compute segment start positions
array[S + 1] int segment_start;
segment_start[1] = 1;
for (s in 1:S) {
segment_start[s + 1] = segment_start[s] + len_theta_ts[s];
}
// Extract sample and positive for nonzero observations
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int sample_nonzero;
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int positive_nonzero;
for (i in 1:N_man_nonzero) {
sample_nonzero[i] = sample[idx_nonzero[i]];
positive_nonzero[i] = positive[idx_nonzero[i]];
}
// Pre-compute family-to-country mapping
array[F] int family_country;
{
for (f in 1:F) {
family_country[f] = 0;
}
for (s in 1:S) {
int f = segment_family[s];
if (family_country[f] == 0) {
family_country[f] = segment_country[s];
}
}
for (f in 1:F) {
if (family_country[f] == 0) {
family_country[f] = 1;
}
}
}
}
parameters {
// =========================================================================
// Latent position parameters - 2 dimensions
// =========================================================================
matrix[2, R] theta_ncp; // Non-centered: [1]=economic_lr, [2]=galtan
matrix[2, S] theta_init_raw; // Initial traits per segment
vector<lower=0>[2] sigma_theta_init; // SD of initial theta per dimension
// =========================================================================
// Three-level hierarchical variance
// =========================================================================
vector[2] mu_sigma_global_raw; // Global mean RW variance
vector<lower=0>[2] tau_sigma_country; // Country deviation scale
matrix[2, P] sigma_country_raw; // Non-centered country deviations
vector<lower=0>[2] tau_sigma_family; // Family deviation scale
matrix[2, F] sigma_family_raw; // Non-centered family deviations
// =========================================================================
// Country-item intercepts (coding convention differences)
// =========================================================================
matrix[P, K_man] country_item_raw;
real<lower=0> sigma_country_item;
// =========================================================================
// Zero-sample parameters
// =========================================================================
real alpha_zs;
vector[T_year] year_effect_raw;
real<lower=0> sigma_year_effect;
vector[S] segment_zs_raw;
real<lower=0> sigma_segment_zs;
vector[N_ciy] ciy_zs_raw;
real<lower=0> sigma_ciy_zs;
// =========================================================================
// Item parameters
// =========================================================================
// Text data: intercept + single positive loading (direction handled in data)
vector[K_man] gamma_man_intercept_raw;
vector<lower=0>[K_man] gamma_man_loading; // Positive loading
// Expert dimension-specific: intercept + slope (direct mapping to dimension)
vector[K_exp_dim] gamma_exp_intercept_raw;
vector<lower=0>[K_exp_dim] gamma_exp_slope;
// Expert general L-R: intercept + slope
vector[K_exp_lr] gamma_lr_intercept_raw;
vector<lower=0>[K_exp_lr] gamma_lr_slope;
// V6: Hierarchical L-R weights (replace simplex[2] lr_weights)
real lr_weight_global; // Global logit-scale weight
vector[P] lr_country_offset_raw; // Country offsets (non-centered)
vector[K_exp_lr] lr_source_offset_raw; // Source offsets (non-centered)
vector[D_lr] lr_decade_offset_raw; // Decade offsets (non-centered)
real<lower=0> sigma_lr_country; // SD of country offsets
real<lower=0> sigma_lr_source; // SD of source offsets
real<lower=0> sigma_lr_decade; // SD of decade offsets
// Precision parameters
real<lower=0> phi_exp_dim; // Precision for dimension-specific (now: only measurement noise)
real<lower=0> phi_exp_lr; // Precision for general L-R (now: only measurement noise)
// Scale parameters
real<lower=0> sigma_intercept_man;
real<lower=0> sigma_loading_man;
real<lower=0> sigma_intercept_exp;
real<lower=0> sigma_slope_exp;
real<lower=0> sigma_intercept_lr;
real<lower=0> sigma_slope_lr;
real mu_lambda_man;
real mu_lambda_exp_dim;
real mu_lambda_exp_lr;
}
transformed parameters {
matrix[2, R] theta; // [1]=economic_lr, [2]=galtan
matrix[2, S] theta_init;
// Three-level hierarchical variance (2D)
matrix[2, P] sigma_theta_country;
for (d in 1:2) {
for (p in 1:P) {
sigma_theta_country[d, p] = log1p_exp(
mu_sigma_global_raw[d] + tau_sigma_country[d] * sigma_country_raw[d, p]
);
}
}
matrix[2, F] sigma_theta_family;
for (d in 1:2) {
for (f in 1:F) {
int c = family_country[f];
sigma_theta_family[d, f] = log1p_exp(
log(sigma_theta_country[d, c]) + tau_sigma_family[d] * sigma_family_raw[d, f]
);
}
}
// Non-centered parameterization for theta_init
for (dim in 1:2) {
theta_init[dim, :] = sigma_theta_init[dim] * theta_init_raw[dim, :];
}
// Country-item intercepts
matrix[P, K_man] country_item_intercept = sigma_country_item * country_item_raw;
// Zero-sample components
vector[T_year] year_effect = sigma_year_effect * year_effect_raw;
vector[S] segment_zs = sigma_segment_zs * segment_zs_raw;
vector[N_ciy] ciy_zs = sigma_ciy_zs * ciy_zs_raw;
// Zero-sample probability
vector[N_man] zero_sample_logit = alpha_zs +
year_effect[year_for_man] +
segment_zs[ss_man] +
ciy_zs[ciy_idx];
vector[N_man] zero_sample_prob = inv_logit(zero_sample_logit);
zero_sample_prob = fmax(fmin(zero_sample_prob, one_minus_eps), eps);
// Construct theta using family-specific random walk variance (2D)
for (dim in 1:2) {
for (s in 1:S) {
int start = segment_start[s];
int Ts = len_theta_ts[s];
int fam = segment_family[s];
real sigma_s = sigma_theta_family[dim, fam];
theta[dim, start] = theta_init[dim, s] + sigma_s * theta_ncp[dim, start];
if (Ts > 1) {
theta[dim, start + 1 : start + Ts - 1] = theta[dim, start] +
cumulative_sum(sigma_s * theta_ncp[dim, start + 1 : start + Ts - 1]);
}
}
}
// Item parameters
vector[K_man] gamma_man_intercept = mu_lambda_man + sigma_intercept_man * gamma_man_intercept_raw;
vector[K_exp_dim] gamma_exp_intercept = mu_lambda_exp_dim + sigma_intercept_exp * gamma_exp_intercept_raw;
vector[K_exp_lr] gamma_lr_intercept = mu_lambda_exp_lr + sigma_intercept_lr * gamma_lr_intercept_raw;
}
model {
// =========================================================================
// PRIORS
// =========================================================================
// Three-level hierarchical variance priors (2D)
mu_sigma_global_raw ~ normal(-0.8, 0.5);
tau_sigma_country ~ normal(0, 0.3);
to_vector(sigma_country_raw) ~ std_normal();
tau_sigma_family ~ normal(0, 0.2);
to_vector(sigma_family_raw) ~ std_normal();
// Other variance priors
sigma_theta_init ~ normal(0, 0.5);
to_vector(theta_init_raw) ~ std_normal();
to_vector(theta_ncp) ~ std_normal();
// Country-item intercept priors
to_vector(country_item_raw) ~ std_normal();
sigma_country_item ~ normal(0, 0.3);
// Zero-sample priors
alpha_zs ~ normal(-1, 1);
year_effect_raw ~ std_normal();
sigma_year_effect ~ normal(0, 0.5);
segment_zs_raw ~ std_normal();
sigma_segment_zs ~ normal(0, 0.3);
ciy_zs_raw ~ std_normal();
sigma_ciy_zs ~ normal(0, 0.3);
// =========================================================================
// Item parameter priors
// =========================================================================
// Text data item priors
mu_lambda_man ~ normal(mn_resp_log_man, 0.5);
sigma_intercept_man ~ normal(0, 1);
sigma_loading_man ~ normal(0, 0.5);
gamma_man_intercept_raw ~ std_normal();
gamma_man_loading ~ normal(1.0, sigma_loading_man);
// Expert dimension item priors
mu_lambda_exp_dim ~ normal(mn_resp_log_exp_dim, 0.5);
sigma_intercept_exp ~ normal(0, 1);
sigma_slope_exp ~ normal(0, 0.5);
gamma_exp_intercept_raw ~ std_normal();
gamma_exp_slope ~ normal(1.0, sigma_slope_exp);
// Expert L-R item priors
mu_lambda_exp_lr ~ normal(mn_resp_log_exp_lr, 0.5);
sigma_intercept_lr ~ normal(0, 1);
sigma_slope_lr ~ normal(0, 0.5);
gamma_lr_intercept_raw ~ std_normal();
gamma_lr_slope ~ normal(1.0, sigma_slope_lr);
// V6: Hierarchical L-R weight priors
lr_weight_global ~ normal(0, 1);
lr_country_offset_raw ~ std_normal();
lr_source_offset_raw ~ std_normal();
lr_decade_offset_raw ~ std_normal();
sigma_lr_country ~ normal(0, 0.5);
sigma_lr_source ~ normal(0, 0.5);
sigma_lr_decade ~ normal(0, 0.5);
// Expert data precision priors
phi_exp_dim ~ gamma(50, 0.5);
phi_exp_lr ~ gamma(10, 0.5);
// =========================================================================
// CDU ANCHOR CONSTRAINT
// With unions: anchors CDU (1375) at moderate center-right position
// Without unions: anchors CDU/CSU (211) as before
// =========================================================================
target += normal_lpdf(theta_init[1, anchor_segment] | 0.2, 0.2); // economic_lr
target += normal_lpdf(theta_init[2, anchor_segment] | 0.2, 0.2); // galtan
// =========================================================================
// LIKELIHOOD 1: Text data (binomial with zero-inflation)
// V4: Mean-constituent averaging for union observations
// =========================================================================
// Zero-sample observations
if (N_man_zero > 0) {
target += sum(log(zero_sample_prob[idx_zero]));
}
// Non-zero observations with constituent averaging
if (N_man_nonzero > 0) {
vector[N_man_nonzero] lin_man;
for (i in 1:N_man_nonzero) {
int nc = n_const_nonzero[i];
int off = const_offset_nonzero[i];
int dim = dim_idx_nonzero[i];
real avg_pos;
if (nc == 1) {
// Fast path: single party (>95% of observations)
avg_pos = theta[dim, const_rr_man[off]];
} else {
// Union: average over constituent thetas
avg_pos = 0;
for (c in 0:(nc-1)) {
avg_pos += theta[dim, const_rr_man[off + c]];
}
avg_pos /= nc;
}
lin_man[i] = gamma_man_intercept[kk_man_nonzero[i]] +
direction_nonzero[i] * gamma_man_loading[kk_man_nonzero[i]] * avg_pos +
country_item_intercept[pp_man_nonzero[i], kk_man_nonzero[i]];
}
for (i in 1:N_man_nonzero) {
target += log1m(zero_sample_prob[orig_idx_nonzero[i]]) +
binomial_logit_lpmf(positive_nonzero[i] | sample_nonzero[i], lin_man[i]);
}
}
// =========================================================================
// LIKELIHOOD 2: Expert dimension-specific data (V5: beta-binomial likelihood)
// V4: Constituent averaging for union-level expert obs
// =========================================================================
{
vector[N_exp_dim] pos;
for (n in 1:N_exp_dim) {
int nc = n_const_exp_dim[n];
int off = const_offset_exp_dim[n];
int dim = dim_idx_exp[n];
if (nc == 1) {
pos[n] = theta[dim, const_rr_exp_dim[off]];
} else {
pos[n] = 0;
for (c in 0:(nc-1)) {
pos[n] += theta[dim, const_rr_exp_dim[off + c]];
}
pos[n] /= nc;
}
}
vector[N_exp_dim] lin_exp_dim;
for (n in 1:N_exp_dim) {
lin_exp_dim[n] = gamma_exp_intercept[kk_exp_dim[n]] +
gamma_exp_slope[kk_exp_dim[n]] * pos[n];
}
vector[N_exp_dim] mu_exp_dim = inv_logit(lin_exp_dim);
mu_exp_dim = fmax(fmin(mu_exp_dim, one_minus_eps), eps);
// K-scaling: phi * K corrects for independent expert perceptions
vector[N_exp_dim] alpha_exp_dim = phi_exp_dim * to_vector(n_experts_exp_dim) .* mu_exp_dim;
vector[N_exp_dim] beta_exp_dim = phi_exp_dim * to_vector(n_experts_exp_dim) .* (1 - mu_exp_dim);
val_dim_int ~ beta_binomial(n_total_exp_dim, alpha_exp_dim, beta_exp_dim);
}
// =========================================================================
// LIKELIHOOD 3: Expert general L-R data (V6: hierarchical per-obs weights)
// V4: Constituent averaging + weighted combination of both dimensions
// V6: Per-observation weights via country + source + decade offsets
// =========================================================================
{
// V6: Compute per-observation economic weight on logit scale
vector[N_exp_lr] logit_w;
for (n in 1:N_exp_lr) {
logit_w[n] = lr_weight_global
+ sigma_lr_country * lr_country_offset_raw[pp_exp_lr[n]]
+ sigma_lr_source * lr_source_offset_raw[kk_exp_lr[n]]
+ sigma_lr_decade * lr_decade_offset_raw[dd_exp_lr[n]];
}
vector[N_exp_lr] w_econ = inv_logit(logit_w);
vector[N_exp_lr] combined_pos;
for (n in 1:N_exp_lr) {
int nc = n_const_exp_lr[n];
int off = const_offset_exp_lr[n];
if (nc == 1) {
int r = const_rr_exp_lr[off];
combined_pos[n] = w_econ[n] * theta[1, r] + (1 - w_econ[n]) * theta[2, r];
} else {
// Average the combined position across constituents
combined_pos[n] = 0;
for (c in 0:(nc-1)) {
int r = const_rr_exp_lr[off + c];
combined_pos[n] += w_econ[n] * theta[1, r] + (1 - w_econ[n]) * theta[2, r];
}
combined_pos[n] /= nc;
}
}
vector[N_exp_lr] lin_exp_lr;
for (n in 1:N_exp_lr) {
lin_exp_lr[n] = gamma_lr_intercept[kk_exp_lr[n]] +
gamma_lr_slope[kk_exp_lr[n]] * combined_pos[n];
}
vector[N_exp_lr] mu_exp_lr = inv_logit(lin_exp_lr);
mu_exp_lr = fmax(fmin(mu_exp_lr, one_minus_eps), eps);
// K-scaling: phi * K corrects for independent expert perceptions
vector[N_exp_lr] alpha_exp_lr = phi_exp_lr * to_vector(n_experts_exp_lr) .* mu_exp_lr;
vector[N_exp_lr] beta_exp_lr = phi_exp_lr * to_vector(n_experts_exp_lr) .* (1 - mu_exp_lr);
val_lr_int ~ beta_binomial(n_total_exp_lr, alpha_exp_lr, beta_exp_lr);
}
}
generated quantities {
// =========================================================================
// Direct outputs (no derivation needed)
// =========================================================================
// Two bipolar scales on [0,1] via inv_logit
vector[R] economic_lr = inv_logit(to_vector(theta[1, :])); // 0=left, 1=right
vector[R] galtan = inv_logit(to_vector(theta[2, :])); // 0=GAL, 1=TAN
// General left-right combining both dimensions (using global weight only)
// Per-R country/source/decade info not available in GQ, so use global mean
real lr_w_econ_global = inv_logit(lr_weight_global);
vector[R] general_lr = inv_logit(
lr_w_econ_global * to_vector(theta[1, :]) + (1 - lr_w_econ_global) * to_vector(theta[2, :])
);
// V6: Expose hierarchical L-R weight diagnostics
real lr_weight_econ_global = lr_w_econ_global;
real lr_sigma_country = sigma_lr_country;
real lr_sigma_source = sigma_lr_source;
real lr_sigma_decade = sigma_lr_decade;
// Expose hierarchical variance diagnostics (2D)
vector[2] mean_sigma_global;
vector[2] mean_sigma_country;
vector[2] mean_sigma_family;
for (d in 1:2) {
mean_sigma_global[d] = log1p_exp(mu_sigma_global_raw[d]);
mean_sigma_country[d] = mean(sigma_theta_country[d, :]);
mean_sigma_family[d] = mean(sigma_theta_family[d, :]);
}
}
+63
View File
@@ -0,0 +1,63 @@
#!/usr/bin/env bash
set -euo pipefail
MODE="${1:-reuse}"
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)"
cd "$repo_root"
case "$MODE" in
full|reuse|dry-run) ;;
*)
echo "Usage: $0 [full|reuse|dry-run]" >&2
exit 1
;;
esac
export PARTY2D_RAW_DATA_DIR="${PARTY2D_RAW_DATA_DIR:-$repo_root/_local/raw}"
export TMPDIR="${TMPDIR:-$repo_root/_local/tmp}"
mkdir -p "$TMPDIR"
required_model_inputs=(
"data/text_data.csv"
"data/expert.csv"
"data/lr_data.csv"
"data/union_mapping.csv"
"data/party_families.csv"
)
if [ "$MODE" = "dry-run" ]; then
echo "Checking commands..."
command -v bash >/dev/null
command -v julia >/dev/null
echo "Checking key files..."
test -f Project.toml
test -f Manifest.toml
test -f models/stan_model_2dim_v6.stan
for input in "${required_model_inputs[@]}"; do
test -f "$input"
done
echo "Checking shell syntax..."
bash -n scripts/01_prepare_data.sh
bash -n scripts/02_fit_model.sh
bash -n scripts/03_extract_estimates.sh
bash -n scripts/04_enrich_estimates.sh
bash -n scripts/05_validate_estimates.sh
bash -n data-setup/run_data_setup.sh
bash -n data-setup/check_raw_data.sh
echo "Checking Julia project can instantiate without running model code..."
julia --project=. -e 'import Pkg; Pkg.instantiate(); println("Julia project OK")'
echo "Dry run passed. No model fitting was run."
exit 0
fi
bash scripts/01_prepare_data.sh
if [ "$MODE" = "full" ]; then
bash scripts/02_fit_model.sh
else
echo "reuse: skipping Stan model fitting; using latest existing model output"
fi
bash scripts/03_extract_estimates.sh
bash scripts/04_enrich_estimates.sh
bash scripts/05_validate_estimates.sh
+28
View File
@@ -0,0 +1,28 @@
#!/usr/bin/env bash
set -euo pipefail
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P)"
cd "$repo_root"
required_model_inputs=(
"data/text_data.csv"
"data/expert.csv"
"data/lr_data.csv"
"data/union_mapping.csv"
"data/party_families.csv"
)
missing=0
for input in "${required_model_inputs[@]}"; do
if [ ! -s "$input" ]; then
echo "Missing model-ready input: $input" >&2
missing=1
fi
done
if [ "$missing" -ne 0 ]; then
echo "Regenerate model inputs with data-setup/run_data_setup.sh, or restore the included data/ files." >&2
exit 1
fi
echo "Model-ready data inputs are present."
+7
View File
@@ -0,0 +1,7 @@
#!/usr/bin/env bash
set -euo pipefail
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P)"
cd "$repo_root"
julia --project=. src/julia/01_run_model.jl
+7
View File
@@ -0,0 +1,7 @@
#!/usr/bin/env bash
set -euo pipefail
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P)"
cd "$repo_root"
julia --project=. src/julia/02_post_estimation.jl
+7
View File
@@ -0,0 +1,7 @@
#!/usr/bin/env bash
set -euo pipefail
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P)"
cd "$repo_root"
julia --project=. src/julia/02_enrich_output.jl
+9
View File
@@ -0,0 +1,9 @@
#!/usr/bin/env bash
set -euo pipefail
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P)"
cd "$repo_root"
julia --project=. src/julia/validate_convergent.jl
julia --project=. src/julia/validate_uncertainty.jl
julia --project=. src/julia/validate_construct.jl
+42
View File
@@ -0,0 +1,42 @@
#!/usr/bin/env bash
set -euo pipefail
repo_root="$(git rev-parse --show-toplevel)"
cd "$repo_root"
fail() {
printf 'PUBLIC CONTENT CHECK FAILED: %s\n' "$1" >&2
exit 1
}
# The bracketed spellings intentionally prevent this guard from matching its
# own detection patterns.
content_pattern='[Rr][Ee][Vv][Ii][Ee][Ww]([Ee][Rr]|[Ss])?|[Ee][Dd][Ii][Tt][Oo][Rr]([[:space:]_-]*[Cc][Oo][Mm][Mm][Ee][Nn][Tt][Ss]?)?|[Mm][Aa][Jj][Oo][Rr][[:space:]_-]*[Rr][Ee][Vv][Ii][Ss][Ii][Oo][Nn]|[Pp][Oo][Ii][Nn][Tt][[:space:]_-]*[Bb][Yy][[:space:]_-]*[Pp][Oo][Ii][Nn][Tt]|[Rr][Ee][Vv][Ii][Ee][Ww][[:space:]_-]*[Pp][Rr][Oo][Cc][Ee][Ss][Ss]|[Ss][Uu][Bb][Mm][Ii][Ss][Ss][Ii][Oo][Nn][[:space:]_-]*[Dd][Ee][Tt][Aa][Ii][Ll][Ss]'
path_pattern='([Rr][Ee][Vv][Ii][Ee][Ww]|[Rr][Ee][Vv][Ii][Ss][Ii][Oo][Nn]|[Rr][Ee][Ss][Pp][Oo][Nn][Ss][Ee][[:space:]_.-]*[Tt][Oo])'
check_ref() {
local ref="$1"
local hits
hits="$(git grep -n -I -E "$content_pattern" "$ref" -- . ':(exclude)*.pdf' 2>/dev/null || true)"
[[ -z "$hits" ]] || fail "non-public wording found in $ref:\n$hits"
}
check_names() {
local names
names="$(git ls-tree -r --name-only HEAD | grep -E "$path_pattern" || true)"
[[ -z "$names" ]] || fail "non-public-looking tracked path(s):\n$names"
}
check_names
while IFS= read -r commit; do
check_ref "$commit"
done < <(git rev-list --all)
while IFS=$'\t' read -r object subject; do
if printf '%s\n%s\n' "$object" "$subject" | grep -E -q "$content_pattern"; then
fail "non-public wording found in reachable ref or commit subject: $object $subject"
fi
done < <(git for-each-ref --format='%(refname)%09%(subject)'; git log --all --format='%H%x09%s')
printf 'OK: public-content audit passed.\n'
+24
View File
@@ -0,0 +1,24 @@
#!/usr/bin/env bash
set -euo pipefail
repo_root="$(git rev-parse --show-toplevel)"
hook="$repo_root/.git/hooks/pre-push"
if [[ -e "$hook" && ! -f "$hook" ]]; then
printf 'Cannot install guard: %s is not a regular file.\n' "$hook" >&2
exit 1
fi
if [[ -f "$hook" ]] && ! grep -Fq 'scripts/check_public_content.sh' "$hook"; then
printf 'Cannot install guard: existing pre-push hook is not managed by this repository.\n' >&2
exit 1
fi
cat > "$hook" <<'EOF'
#!/usr/bin/env bash
set -euo pipefail
repo_root="$(git rev-parse --show-toplevel)"
exec bash "$repo_root/scripts/check_public_content.sh"
EOF
chmod +x "$hook"
printf 'Installed public-content pre-push guard.\n'
+340
View File
@@ -0,0 +1,340 @@
#!/usr/bin/env julia
#############################################################################
## run_model.jl
## Main runner for latent trait model
## Executes the complete pipeline: data loading → preparation → model fitting
##
## Supports two model versions:
## - "2dim": 2D bipolar model (V1) - estimates economic left-right and cultural cosmopolitan--traditionalist positions directly
## - "4dim": 4D unipolar model (V10) - estimates 4 traits, derives 2 scales
##
## Default is 2D model (better identification, faster convergence)
#############################################################################
using Dates
#############################################################################
## EXECUTION CONFIGURATION (Change these values as needed)
#############################################################################
const MODEL_VERSION = "2dim" # "2dim" (recommended) or "4dim"
const STAN_MODEL_FILE = MODEL_VERSION == "2dim" ? "models/stan_model_2dim_v6.stan" : "models/stan_model_4dim_v10.stan"
const NUM_CHAINS = parse(Int, get(ENV, "PARTY2D_NUM_CHAINS", "4"))
const NUM_WARMUP = parse(Int, get(ENV, "PARTY2D_NUM_WARMUP", "1000"))
const NUM_SAMPLES = parse(Int, get(ENV, "PARTY2D_NUM_SAMPLES", "2000"))
const ADAPT_DELTA = 0.95 # Target acceptance probability
const MAX_DEPTH = 15 # Maximum tree depth
const START_YEAR = 1944 # First year to include (avoid sparse early data)
println("=" ^ 70)
println("Latent Trait Model - Estimation Pipeline")
println("=" ^ 70)
println("Started at: ", Dates.now())
println("Configuration:")
println(" Model version: $(MODEL_VERSION)")
println(" Stan model: $(STAN_MODEL_FILE)")
println(" Chains: $(NUM_CHAINS)")
println(" Warmup iterations: $(NUM_WARMUP)")
println(" Sampling iterations: $(NUM_SAMPLES)")
println(" Total iterations per chain: $(NUM_WARMUP + NUM_SAMPLES)")
println(" Adapt delta: $(ADAPT_DELTA)")
println(" Max depth: $(MAX_DEPTH)")
println(" Start year: $(START_YEAR)")
if MODEL_VERSION == "2dim"
println("\n 2D MODEL: Estimates economic left-right and cultural cosmopolitan--traditionalist positions directly")
println(" (Half the parameters, better convergence)")
else
println("\n 4D MODEL: Estimates 4 traits, derives 2 scales")
println(" (Known identification issues - see VERSION_HISTORY.md)")
end
println("=" ^ 70)
# Include pipeline modules
include("pipeline/00_validation.jl") # Validation checks
include("pipeline/02_data_loading.jl")
include("pipeline/03_data_preparation.jl")
include("pipeline/04_model_execution.jl")
include("pipeline/05_results_processing.jl")
# Load robust save module
include("pipeline/06_save_model.jl")
import .RobustSave: robust_save_model
function run_model(;
num_chains=NUM_CHAINS,
num_warmup=NUM_WARMUP,
num_samples=NUM_SAMPLES,
adapt_delta=ADAPT_DELTA,
max_depth=MAX_DEPTH,
model_file=STAN_MODEL_FILE,
start_year=START_YEAR,
data_dir="data"
)
"""Run the complete latent trait model pipeline (2D or 4D based on MODEL_VERSION)"""
try
# Step 1: Load and preprocess data
println("\n" * "="^50)
println("STEP 1: DATA LOADING")
println("="^50)
manifesto, expert_dim, expert_lr, year0, union_to_constituents, constituent_to_union = load_and_preprocess_4dim_data(start_year; data_dir=data_dir)
# Step 2: Prepare Stan data structure
println("\n" * "="^50)
println("STEP 2: DATA PREPARATION")
println("="^50)
# Prepare indices and mappings (V4: union-aware)
data_prep = prepare_4dim_stan_data(manifesto, expert_dim, expert_lr, year0;
union_to_constituents=union_to_constituents,
constituent_to_union=constituent_to_union)
# Finalize Stan data dictionary (V4: includes constituent arrays)
final_data = finalize_4dim_stan_data(
data_prep.manifesto, data_prep.expert_dim, data_prep.expert_lr,
data_prep.segment_year, data_prep.segment_info,
data_prep.all_parties, data_prep.all_groups,
data_prep.group_to_index, year0, data_prep.S, data_prep.J, data_prep.P, data_prep.R,
data_prep.N_ciy, data_prep.len_theta_ts, data_prep.segment_country_idx,
data_prep.F, data_prep.segment_family_idx, data_prep.anchor_segment_idx;
N_const_man_total=data_prep.N_const_man_total,
n_const_man=data_prep.n_const_man,
const_offset_man=data_prep.const_offset_man,
const_rr_man=data_prep.const_rr_man,
N_const_exp_dim_total=data_prep.N_const_exp_dim_total,
n_const_exp_dim=data_prep.n_const_exp_dim,
const_offset_exp_dim=data_prep.const_offset_exp_dim,
const_rr_exp_dim=data_prep.const_rr_exp_dim,
N_const_exp_lr_total=data_prep.N_const_exp_lr_total,
n_const_exp_lr=data_prep.n_const_exp_lr,
const_offset_exp_lr=data_prep.const_offset_exp_lr,
const_rr_exp_lr=data_prep.const_rr_exp_lr
)
dat_4dim = final_data.dat_4dim
# Step 3: Validate data BEFORE running Stan
println("\n" * "="^50)
println("STEP 3: DATA VALIDATION")
println("="^50)
if !validate_stan_data(dat_4dim; verbose=true)
error("Data validation failed - see errors above")
end
estimate_memory_requirements(dat_4dim; verbose=true)
# Step 4: Create initialization function
println("\n" * "="^50)
println("STEP 4: MODEL INITIALIZATION")
println("="^50)
# Determine model version for initialization
model_init_version = MODEL_VERSION == "2dim" ? "v1_2dim" : "v10"
# Use S (segments) for initialization, not J (parties)
init_fn = create_init_function(dat_4dim, data_prep.S, data_prep.P,
data_prep.R, final_data.T_year, data_prep.N_ciy;
model_version=model_init_version)
# Validate initialization values
println("\nValidating initialization for chain 1...")
test_init = init_fn()
if !validate_init_values(test_init; verbose=true)
error("Initialization validation failed - see errors above")
end
# Step 5: Run Stan model
println("\n" * "="^50)
println("STEP 5: MODEL EXECUTION")
println("="^50)
# Create temp folder for output
temp_folder = mktempdir()
println("Temporary folder for Stan output: $temp_folder")
# Run Stan model
stanmodel = run_4dim_stan_model(
dat_4dim, init_fn, temp_folder;
num_chains=num_chains,
num_warmup=num_warmup,
num_samples=num_samples,
adapt_delta=adapt_delta,
max_depth=max_depth,
model_file=model_file
)
# Step 6: Results Processing & Diagnostics
println("\n" * "="^50)
println("STEP 6: RESULTS PROCESSING & DIAGNOSTICS")
println("="^50)
results = extract_model_results_4dim(stanmodel)
diagnostics = compute_model_diagnostics(stanmodel)
println("\nCONVERGENCE DIAGNOSTICS:")
println(" Max R-hat: $(round(diagnostics.max_rhat, digits=4))")
println(" Mean R-hat: $(round(diagnostics.mean_rhat, digits=4))")
println(" High R-hat count: $(diagnostics.high_rhat_count)")
println(" Min ESS: $(round(diagnostics.min_ess, digits=0))")
println(" Mean ESS: $(round(diagnostics.mean_ess, digits=0))")
println(" Convergence Status: $(diagnostics.convergence_status)")
# Step 7: Save results
println("\n" * "="^50)
println("STEP 7: SAVING RESULTS")
println("="^50)
println(" Using save-local-then-move strategy...")
# Prepare data for saving (SINGLE copy of stanmodel, not multiple!)
model_data_to_save = Dict{String, Any}(
# StanModel object (contains all MCMC samples)
"stanmodel_object" => stanmodel,
# Processed diagnostics
"diagnostics_summary" => diagnostics.diagnostics_summary,
"data_dict" => dat_4dim,
# Original data for reference
"manifesto" => final_data.manifesto,
"expert_dim" => final_data.expert_dim,
"expert_lr" => final_data.expert_lr,
"segment_year" => final_data.segment_year, # V10: segment-year mapping
"segment_info" => final_data.segment_info, # V10: segment metadata (party_id, segment_num, year range)
# Metadata with convergence info
"model_info" => Dict(
"timestamp" => Dates.format(Dates.now(), "yyyy-mm-dd_HH-MM-SS"),
"max_rhat" => diagnostics.max_rhat,
"mean_rhat" => diagnostics.mean_rhat,
"min_ess" => diagnostics.min_ess,
"mean_ess" => diagnostics.mean_ess,
"convergence_status" => diagnostics.convergence_status,
"model_file" => model_file,
"model_version" => MODEL_VERSION,
"num_chains" => num_chains,
"num_warmup" => num_warmup,
"num_samples" => num_samples,
"adapt_delta" => adapt_delta,
"max_depth" => max_depth,
"year0" => year0,
"dimensions" => MODEL_VERSION == "2dim" ?
["economic_lr", "galtan"] :
["pro_market", "pro_welfare", "cosmopolitan", "traditional"]
)
)
# Save using CSV-first robust system. Chains have already been secured
# by model execution; robust_save_model adds metadata/data and verifies.
save_dir = data_dir != "data" ? joinpath(data_dir, "model_run") : "outputs/model_outputs"
output_file = robust_save_model(
stanmodel,
model_data_to_save,
save_dir;
compress=true, # Ignored by CSV-first save implementation
keep_local_backups=2 # Ignored; chains are already saved before this step
)
# Robust save module already verified everything!
println("\n" * "="^70)
println("MODEL EXECUTION COMPLETED SUCCESSFULLY!")
println("="^70)
println(" Max R-hat: $(round(diagnostics.max_rhat, digits=4))")
println(" Mean R-hat: $(round(diagnostics.mean_rhat, digits=4))")
println(" Convergence: $(diagnostics.convergence_status)")
println(" Output file: $output_file")
# Print summary statistics
println("\nModel Summary ($(MODEL_VERSION == "2dim" ? "2D Direct Bipolar" : "4D Unipolar")):")
println(" Segments: $(data_prep.S)")
println(" Parties with valid segments: $(data_prep.J)")
println(" Countries: $(data_prep.P)")
println(" Segment-year combinations: $(data_prep.R)")
println(" Years: $(final_data.T_year)")
println(" Manifesto observations: $(dat_4dim["N_man"])")
println(" Expert dimension-specific observations: $(dat_4dim["N_exp_dim"])")
println(" Expert general L-R observations: $(dat_4dim["N_exp_lr"])")
if MODEL_VERSION == "2dim"
println(" Dimensions estimated: 2 (economic left-right, cultural cosmopolitan--traditionalist)")
println(" Theta parameters: $(2 * data_prep.R) (2 × R)")
else
println(" Dimensions estimated: 4 (pro_market, pro_welfare, cosmopolitan, traditional)")
println(" Theta parameters: $(4 * data_prep.R) (4 × R)")
end
println("=" ^ 70)
# Cleanup temp folder after successful save
println("\nCLEANING UP TEMPORARY FILES...")
try
if isdir(temp_folder)
rm(temp_folder, recursive=true, force=true)
println(" Removed temporary folder: $temp_folder")
end
catch cleanup_error
println(" Warning: Could not remove temp folder: $cleanup_error")
println(" (This won't affect your saved results)")
end
return true
catch e
println("\nERROR in model pipeline: $e")
println("Stack trace:")
showerror(stdout, e, catch_backtrace())
rethrow(e)
end
end
function main(args=ARGS)
# Parse --data-dir argument
data_dir = "data"
for (i, arg) in enumerate(args)
if arg == "--data-dir" && i < length(args)
data_dir = args[i + 1]
elseif startswith(arg, "--data-dir=")
data_dir = split(arg, "=", limit=2)[2]
end
end
if data_dir != "."
println("Using data directory: $data_dir")
end
println("Executing $(MODEL_VERSION) latent trait model pipeline...")
# Check that required data files exist (in data_dir)
required_data = [joinpath(data_dir, f) for f in ["text_data.csv", "expert.csv", "lr_data.csv"]]
required_files = vcat(required_data, [STAN_MODEL_FILE])
missing_files = []
for file in required_files
if !isfile(file)
push!(missing_files, file)
end
end
if !isempty(missing_files)
println("ERROR: Missing required files:")
for file in missing_files
println(" - $file")
end
if any(f -> endswith(f, "text_data.csv") || endswith(f, "expert.csv") || endswith(f, "lr_data.csv"), missing_files)
println("\nTo generate data files, run:")
println(" bash scripts/01_prepare_data.sh")
end
error("Cannot proceed without required files")
end
# Run the complete pipeline
results = run_model(data_dir=data_dir)
println("\n$(MODEL_VERSION) latent trait model pipeline completed successfully!")
println("Check outputs/model_outputs/latest/ for chain CSVs and metadata.")
end
# Main execution
if abspath(PROGRAM_FILE) == @__FILE__
main()
end
+138
View File
@@ -0,0 +1,138 @@
#!/usr/bin/env julia
#=
02_enrich_output.jl - Enrich party positions CSV with model-input-derived metadata
Fast post-processing script that adds union membership status from the model-ready
inputs. Operates purely on CSV files (no chain loading). Takes seconds, not minutes.
Usage:
julia 02_enrich_output.jl # enriches latest party_positions_*.csv
julia 02_enrich_output.jl somefile.csv # enriches a specific file
Adds columns:
in_union - 1 if party's union had a joint manifesto that year, 0 otherwise
=#
using CSV
using DataFrames
using Dates
function find_latest_output()
outdir = "outputs/estimations/latest"
files = filter(f -> startswith(f, "party_positions_") && endswith(f, ".csv") &&
!contains(f, "metadata") && !contains(f, "tables"),
readdir(outdir))
isempty(files) && error("No party_positions_*.csv found in $outdir/")
sort!(files, rev=true)
return joinpath(outdir, files[1])
end
function enrich(input_file::String)
println("="^60)
println("ENRICH OUTPUT")
println("="^60)
println("Input: $input_file")
output = CSV.read(input_file, DataFrame)
println(" Rows: $(nrow(output)), Columns: $(ncol(output))")
# --- Union mapping (for in_union) ---
union_mapping_file = joinpath("data", "union_mapping.csv")
constituent_to_union = Dict{Int, Int}()
if isfile(union_mapping_file)
union_df = CSV.read(union_mapping_file, DataFrame)
for row in eachrow(union_df)
constituent_to_union[row.expert_pf_id] = row.manifesto_pf_id
end
end
# --- in_union dummy (year-varying) ---
text_data_file = "data/text_data.csv"
output.in_union = zeros(Int, nrow(output))
if isfile(text_data_file) && !isempty(constituent_to_union)
text_df = CSV.read(text_data_file, DataFrame)
# Build set of (party_pf_id, year) pairs for manifesto data
manifesto_text = filter(r -> r.project == "Manifesto Project", text_df)
manifesto_party_years = Set{Tuple{Int, Int}}()
for row in eachrow(manifesto_text)
push!(manifesto_party_years, (row.party, row.year))
end
# Process each (party, segment) group
gdf = groupby(output, [:party_id, :segment_num])
for subdf in gdf
pid = subdf.party_id[1]
!haskey(constituent_to_union, pid) && continue
union_id = constituent_to_union[pid]
# Get row indices in the full output for this group
idxs = parentindices(subdf)[1]
# Determine in_union at election years
election_year_vals = Dict{Int, Int}()
for (j, row) in enumerate(eachrow(subdf))
if (union_id, row.year) in manifesto_party_years
election_year_vals[row.year] = 1
elseif (pid, row.year) in manifesto_party_years
election_year_vals[row.year] = 0
end
end
# Forward-fill within segment
sorted_pairs = sort(collect(zip(subdf.year, idxs)))
last_val = 0
for (yr, idx) in sorted_pairs
if haskey(election_year_vals, yr)
last_val = election_year_vals[yr]
end
output.in_union[idx] = last_val
end
end
n_in_union = count(x -> x == 1, output.in_union)
println(" in_union: $n_in_union rows flagged as union members")
else
println(" WARNING: Could not compute in_union (missing files)")
end
# --- Reorder columns ---
estimate_cols = Symbol[]
for base in ["economic_lr", "galtan", "pro_market", "pro_welfare", "cosmopolitan", "traditional"]
sym = Symbol(base)
if hasproperty(output, sym)
push!(estimate_cols, sym)
push!(estimate_cols, Symbol("$(base)_se"))
push!(estimate_cols, Symbol("$(base)_q025"))
push!(estimate_cols, Symbol("$(base)_q975"))
end
end
# Fix country code: MO (Macau) → MK (North Macedonia) — GPS uses wrong ISO2
output.country = replace(output.country, "MO" => "MK")
col_order = vcat(
[:party_id, :country, :year, :segment_num, :union_party_id, :in_union],
estimate_cols
)
col_order = filter(c -> hasproperty(output, c), col_order)
select!(output, col_order)
# --- Write back ---
CSV.write(input_file, output)
println("\n Wrote: $input_file")
println(" Columns ($(ncol(output))): $(join(string.(names(output)), ", "))")
return output
end
function main(args=ARGS)
input = length(args) >= 1 ? args[1] : find_latest_output()
enrich(input)
println("\nDone.")
end
if abspath(PROGRAM_FILE) == @__FILE__
main()
end
+986
View File
@@ -0,0 +1,986 @@
#!/usr/bin/env julia
#=
02_post_estimation.jl - Extract party position estimates from Stan model output
Supports both model versions:
- 2D model (V1): Extracts economic left-right and cultural cosmopolitan--traditionalist positions directly
- 4D model (V10): Extracts 4 traits + 2 derived scales
V10/V1 UPDATE: Handles segment-based indexing
- Maps segment results back to original party IDs
- Adds segment_num column to indicate which segment of the party
- Flags discontinuities for parties with multiple segments
This script:
1. Auto-detects the latest model run in model_outputs/
2. Loads the chain CSV files and data mappings
3. Detects model version from metadata or column names
4. Extracts posterior summaries for all segment-year positions
5. Maps Stan parameter indices back to real party IDs, segment numbers, and years
6. Saves output as wide-format CSV with uncertainty estimates
Usage:
julia 02_post_estimation.jl
Output:
party_positions_YYYY-MM-DD_HH-MM-SS.csv
=#
using CSV
using DataFrames
using Statistics
using JSON
using Dates
using Printf
# =============================================================================
# STEP 0: Auto-detect latest run
# =============================================================================
function find_latest_run(base_dir::String="outputs/model_outputs/latest")
if !isdir(base_dir)
error("Model outputs directory not found: $base_dir")
end
runs = filter(d -> startswith(d, "run_") && isdir(joinpath(base_dir, d)), readdir(base_dir))
if isempty(runs)
error("No runs found in $base_dir")
end
# Sort by timestamp in directory name (format: run_YYYY-MM-DD_HH-MM-SS)
sort!(runs, rev=true)
latest = joinpath(base_dir, runs[1])
println("Found $(length(runs)) run(s). Using latest: $latest")
return latest
end
# =============================================================================
# STEP 1: Load data and build segment-year lookup
# =============================================================================
function load_run_data(run_dir::String)
println("\n" * "="^60)
println("LOADING RUN DATA")
println("="^60)
data_dir = joinpath(run_dir, "data")
chains_dir = joinpath(run_dir, "chains")
# Check required files exist
required_files = [
joinpath(data_dir, "text_data.csv"),
joinpath(data_dir, "expert_dim.csv"),
joinpath(data_dir, "expert_lr.csv"),
joinpath(run_dir, "metadata.json")
]
for f in required_files
if !isfile(f)
error("Required file not found: $f")
end
end
# Load data files
println("Loading text_data.csv...")
text_data = CSV.read(joinpath(data_dir, "text_data.csv"), DataFrame)
println(" Rows: $(nrow(text_data))")
println("Loading expert_dim.csv...")
expert_dim = CSV.read(joinpath(data_dir, "expert_dim.csv"), DataFrame)
println(" Rows: $(nrow(expert_dim))")
println("Loading expert_lr.csv...")
expert_lr = CSV.read(joinpath(data_dir, "expert_lr.csv"), DataFrame)
println(" Rows: $(nrow(expert_lr))")
println("Loading metadata.json...")
metadata = JSON.parsefile(joinpath(run_dir, "metadata.json"))
println(" year0: $(metadata["year0"])")
println(" Model: $(metadata["model_file"])")
# V10: Load segment_info if available
segment_info_file = joinpath(data_dir, "segment_info.csv")
segment_info = nothing
if isfile(segment_info_file)
println("Loading segment_info.csv (V10)...")
segment_info = CSV.read(segment_info_file, DataFrame)
println(" Segments: $(nrow(segment_info))")
end
# V10: Load segment_year_map if available
segment_year_file = joinpath(data_dir, "segment_year_map.csv")
segment_year_map = nothing
if isfile(segment_year_file)
println("Loading segment_year_map.csv (V10)...")
segment_year_map = CSV.read(segment_year_file, DataFrame)
println(" Segment-years: $(nrow(segment_year_map))")
end
# Find chain files
chain_files = filter(f -> endswith(f, ".csv") && startswith(f, "chain_"), readdir(chains_dir))
println("\nFound $(length(chain_files)) chain file(s)")
return (
text_data = text_data,
expert_dim = expert_dim,
expert_lr = expert_lr,
metadata = metadata,
segment_info = segment_info,
segment_year_map = segment_year_map,
chain_files = [joinpath(chains_dir, f) for f in sort(chain_files)],
run_dir = run_dir
)
end
function normalize_country_value(value)
if ismissing(value)
return missing
end
txt = strip(string(value))
return isempty(txt) ? missing : txt
end
function build_party_country_map(text_data::DataFrame, expert_dim::DataFrame, expert_lr::DataFrame)
merged = unique(vcat(
select(text_data, :party, :country),
select(expert_dim, :party, :country),
select(expert_lr, :party, :country)
))
party_to_country = Dict{Int, String}()
for row in eachrow(merged)
pid = tryparse(Int, string(row.party))
if pid === nothing
continue
end
c = normalize_country_value(row.country)
if !ismissing(c)
party_to_country[pid] = c
end
end
return party_to_country
end
function load_constituent_to_union_map()::Dict{Int, Int}
mapping_file = joinpath("data", "union_mapping.csv")
constituent_to_union = Dict{Int, Int}()
if isfile(mapping_file)
union_df = CSV.read(mapping_file, DataFrame)
for row in eachrow(union_df)
constituent_to_union[row.expert_pf_id] = row.manifesto_pf_id
end
end
return constituent_to_union
end
function resolve_party_country(pid_value,
party_to_country::Dict{Int, String},
constituent_to_union::Dict{Int, Int})
pid = tryparse(Int, string(pid_value))
if pid === nothing
return missing, "unresolved"
end
if haskey(party_to_country, pid)
return party_to_country[pid], "direct"
end
if haskey(constituent_to_union, pid)
uid = constituent_to_union[pid]
if haskey(party_to_country, uid)
return party_to_country[uid], "union_fallback"
end
end
return missing, "unresolved"
end
function apply_country_resolution!(df::DataFrame,
party_col::Symbol,
country_col::Symbol,
party_to_country::Dict{Int, String},
constituent_to_union::Dict{Int, Int})
resolved_country = Union{Missing, String}[]
source_counts = Dict("direct" => 0, "union_fallback" => 0, "unresolved" => 0)
unresolved_parties = Set{Int}()
for pid in df[!, party_col]
country, source = resolve_party_country(pid, party_to_country, constituent_to_union)
push!(resolved_country, country)
source_counts[source] += 1
if source == "unresolved"
pid_int = tryparse(Int, string(pid))
if pid_int !== nothing
push!(unresolved_parties, pid_int)
end
end
end
df[!, country_col] = resolved_country
return source_counts, sort!(collect(unresolved_parties))
end
function fill_missing_countries!(df::DataFrame,
segment_info::Union{DataFrame, Nothing},
party_to_country::Dict{Int, String},
constituent_to_union::Dict{Int, Int})
if !hasproperty(df, :country)
source_counts, unresolved = apply_country_resolution!(
df, :party_id, :country, party_to_country, constituent_to_union
)
return source_counts, unresolved
end
normalized = Union{Missing, String}[]
for val in df.country
push!(normalized, normalize_country_value(val))
end
df.country = normalized
source_counts = Dict("direct" => 0, "union_fallback" => 0, "segment_info" => 0, "unresolved" => 0)
segment_country_by_id = Dict{Int, String}()
if segment_info !== nothing && hasproperty(segment_info, :country)
for row in eachrow(segment_info)
c = normalize_country_value(row.country)
if !ismissing(c)
segment_country_by_id[Int(row.segment_id)] = c
end
end
end
unresolved_parties = Set{Int}()
for i in 1:nrow(df)
if !ismissing(df.country[i])
continue
end
if hasproperty(df, :segment_id) && haskey(segment_country_by_id, Int(df.segment_id[i]))
df.country[i] = segment_country_by_id[Int(df.segment_id[i])]
source_counts["segment_info"] += 1
continue
end
country, source = resolve_party_country(df.party_id[i], party_to_country, constituent_to_union)
if !ismissing(country)
df.country[i] = country
source_counts[source] += 1
else
source_counts["unresolved"] += 1
pid_int = tryparse(Int, string(df.party_id[i]))
pid_int !== nothing && push!(unresolved_parties, pid_int)
end
end
return source_counts, sort!(collect(unresolved_parties))
end
# =============================================================================
# STEP 2: Build complete segment-year mapping (V10) or party-year mapping (V9)
# =============================================================================
function build_segment_year_map(text_data::DataFrame, expert_dim::DataFrame, expert_lr::DataFrame,
segment_info::Union{DataFrame, Nothing},
segment_year_map::Union{DataFrame, Nothing},
run_dir::String,
year0::Int)
println("\n" * "="^60)
println("BUILDING SEGMENT-YEAR MAPPING")
println("="^60)
party_to_country = build_party_country_map(text_data, expert_dim, expert_lr)
constituent_to_union = load_constituent_to_union_map()
# V10: Use segment_year_map if available
if segment_year_map !== nothing && segment_info !== nothing
println("Using segment_year_map.csv (V10 mode)")
# Convert relative Year to absolute year
if hasproperty(segment_year_map, :Year)
segment_year_map.year = segment_year_map.Year .+ year0
elseif !hasproperty(segment_year_map, :year)
error("segment_year_map has no Year or year column")
end
# Add party_id from segment_info if not already present
if !hasproperty(segment_year_map, :party_id)
segment_id_to_party = Dict(row.segment_id => row.party_id for row in eachrow(segment_info))
segment_year_map.party_id = [segment_id_to_party[sid] for sid in segment_year_map.segment_id]
end
# Add segment_num from segment_info if not already present
if !hasproperty(segment_year_map, :segment_num)
segment_id_to_segnum = Dict(row.segment_id => row.segment_num for row in eachrow(segment_info))
segment_year_map.segment_num = [segment_id_to_segnum[sid] for sid in segment_year_map.segment_id]
end
# Resolve/fill country column using segment metadata first, then direct and union-fallback lookup.
source_counts, unresolved = fill_missing_countries!(
segment_year_map, segment_info, party_to_country, constituent_to_union
)
direct_count = get(source_counts, "direct", 0)
union_count = get(source_counts, "union_fallback", 0)
segment_info_count = get(source_counts, "segment_info", 0)
unresolved_count = count(ismissing, segment_year_map.country)
println(" Country resolution fill counts: direct=$direct_count, union_fallback=$union_count, segment_info=$segment_info_count, unresolved_rows=$unresolved_count")
if !isempty(unresolved)
println(" Warning: unresolved country party IDs (first 20): $(unresolved[1:min(20, length(unresolved))])")
end
R = maximum(segment_year_map.rr)
n_segments = length(unique(segment_year_map.segment_id))
n_parties = length(unique(segment_year_map.party_id))
println("Loaded segment_year_map: $(nrow(segment_year_map)) segment-years (R=$R)")
println(" Unique segments: $n_segments")
println(" Unique parties: $n_parties")
# Count observed vs interpolated
observed_rrs = Set{Int}()
if hasproperty(text_data, :rr_man)
union!(observed_rrs, Set(text_data.rr_man))
end
if hasproperty(expert_dim, :rr_exp_dim)
union!(observed_rrs, Set(expert_dim.rr_exp_dim))
end
if hasproperty(expert_lr, :rr_exp_lr)
union!(observed_rrs, Set(expert_lr.rr_exp_lr))
end
n_observed = length(intersect(Set(segment_year_map.rr), observed_rrs))
n_interpolated = nrow(segment_year_map) - n_observed
println(" Observed segment-years: $n_observed")
println(" Interpolated segment-years: $n_interpolated")
return segment_year_map, R, segment_info
end
# V9 fallback: Use party_year_map
party_year_file = joinpath(run_dir, "data", "party_year_map.csv")
if isfile(party_year_file)
println("Loading party_year_map.csv (V9 fallback mode)")
party_year_map = CSV.read(party_year_file, DataFrame)
# Add party_id column (same as party for V9)
if !hasproperty(party_year_map, :party_id)
party_year_map.party_id = party_year_map.party
end
# Add segment_num column (always 1 for V9)
if !hasproperty(party_year_map, :segment_num)
party_year_map.segment_num = ones(Int, nrow(party_year_map))
end
# Add segment_id column (same as party index for V9)
if !hasproperty(party_year_map, :segment_id)
party_year_map.segment_id = party_year_map.party
end
# Convert relative Year to absolute year
if hasproperty(party_year_map, :Year)
party_year_map.year = party_year_map.Year .+ year0
elseif !hasproperty(party_year_map, :year)
error("party_year_map has no Year or year column")
end
# Resolve/fill country column
if !hasproperty(party_year_map, :country)
source_counts, unresolved = apply_country_resolution!(
party_year_map, :party_id, :country, party_to_country, constituent_to_union
)
direct_count = source_counts["direct"]
union_count = source_counts["union_fallback"]
unresolved_count = source_counts["unresolved"]
println(" Country resolution sources: direct=$direct_count, union_fallback=$union_count, unresolved=$unresolved_count")
if !isempty(unresolved)
println(" Warning: unresolved country party IDs (first 20): $(unresolved[1:min(20, length(unresolved))])")
end
else
normalized = Union{Missing, String}[]
for val in party_year_map.country
push!(normalized, normalize_country_value(val))
end
party_year_map.country = normalized
end
R = maximum(party_year_map.rr)
println("Loaded party_year_map: $(nrow(party_year_map)) party-years (R=$R)")
return party_year_map, R, nothing
end
# Fallback: Reconstruct from data files
@warn "No mapping file found, reconstructing from data (observed years only)"
# Extract unique party-year-rr combinations from text_data
text_map = unique(select(text_data, :party, :country, :year, :rr_man))
rename!(text_map, :rr_man => :rr)
text_map.party_id = text_map.party
text_map.segment_num = ones(Int, nrow(text_map))
expert_dim_map = unique(select(expert_dim, :party, :country, :year, :rr_exp_dim))
rename!(expert_dim_map, :rr_exp_dim => :rr)
expert_dim_map.party_id = expert_dim_map.party
expert_dim_map.segment_num = ones(Int, nrow(expert_dim_map))
expert_lr_map = unique(select(expert_lr, :party, :country, :year, :rr_exp_lr))
rename!(expert_lr_map, :rr_exp_lr => :rr)
expert_lr_map.party_id = expert_lr_map.party
expert_lr_map.segment_num = ones(Int, nrow(expert_lr_map))
combined = vcat(text_map, expert_dim_map, expert_lr_map)
segment_year_map = unique(combined)
sort!(segment_year_map, :rr)
R = maximum(segment_year_map.rr)
println("Reconstructed mapping: $(nrow(segment_year_map)) segment-years (R=$R)")
return segment_year_map, R, nothing
end
# =============================================================================
# STEP 3: Load and combine chains
# =============================================================================
function load_chains(chain_files::Vector{String})
println("\n" * "="^60)
println("LOADING STAN CHAINS")
println("="^60)
flush(stdout)
chains = DataFrame[]
# The full Stan CSVs are very wide (hundreds of thousands of columns). For
# post-estimation we only need party-position generated quantities. Reading
# all columns can take hours and allocate many GB of irrelevant parameters.
post_estimation_prefixes = (
"economic_lr.",
"galtan.",
"pro_market.",
"pro_welfare.",
"cosmopolitan.",
"traditional.",
)
keep_post_estimation_col(_i, name) = any(startswith(String(name), p) for p in post_estimation_prefixes)
for (i, f) in enumerate(chain_files)
println("Loading chain $i: $(basename(f))...")
flush(stdout)
# Skip comment lines (Stan header) and parse only needed quantities.
chain = CSV.read(f, DataFrame; comment="#", select=keep_post_estimation_col)
println(" Samples: $(nrow(chain)), Parameters: $(ncol(chain))")
flush(stdout)
push!(chains, chain)
end
# Combine chains
println("Combining selected chain columns...")
flush(stdout)
combined = vcat(chains...)
println("\nCombined: $(nrow(combined)) total samples")
println("Selected parameters: $(ncol(combined))")
flush(stdout)
return combined
end
# =============================================================================
# STEP 4: Extract generated quantities
# =============================================================================
"""
Detect model version from chain column names.
Returns "2dim" or "4dim".
"""
function detect_model_version(chains::DataFrame)
cols = names(chains)
# 2D model has economic_lr but NOT pro_market
has_economic_lr = any(c -> startswith(string(c), "economic_lr."), cols)
has_pro_market = any(c -> startswith(string(c), "pro_market."), cols)
if has_economic_lr && !has_pro_market
return "2dim"
elseif has_pro_market
return "4dim"
else
error("Could not detect model version from chain columns")
end
end
function extract_estimates(chains::DataFrame, segment_year_map::DataFrame, R::Int)
println("\n" * "="^60)
println("EXTRACTING POSTERIOR ESTIMATES")
println("="^60)
# Auto-detect model version from columns
model_version = detect_model_version(chains)
println("Detected model version: $model_version")
# Select quantities based on model version
if model_version == "2dim"
# 2D model: economic left-right and cultural cosmopolitan--traditionalist positions are directly estimated
# (general_lr is computed in Stan for anchoring but not extracted as output)
quantities = ["economic_lr", "galtan"]
test_col = "economic_lr.1"
else
# 4D model: 4 traits + 2 derived scales
quantities = ["pro_market", "pro_welfare", "cosmopolitan", "traditional", "economic_lr", "galtan"]
test_col = "pro_market.1"
end
# Check that columns exist
if !hasproperty(chains, Symbol(test_col))
error("Column $test_col not found in chains. Available columns: $(first(names(chains), 10))...")
end
n_samples = nrow(chains)
println("Samples per parameter: $n_samples")
# Load union mapping for adding union_party_id column
union_mapping_file = joinpath("data", "union_mapping.csv")
constituent_to_union_pf = Dict{Int, Int}()
if isfile(union_mapping_file)
union_df = CSV.read(union_mapping_file, DataFrame)
for row in eachrow(union_df)
constituent_to_union_pf[row.expert_pf_id] = row.manifesto_pf_id
end
end
# Pre-allocate output DataFrame
n_rows = nrow(segment_year_map)
# Add union_party_id column: NA for standalone parties, union PF ID for constituents
union_ids = Union{Int, Missing}[]
for pid in segment_year_map.party_id
pid_int = isa(pid, Integer) ? pid : tryparse(Int, string(pid))
if pid_int !== nothing && haskey(constituent_to_union_pf, pid_int)
push!(union_ids, constituent_to_union_pf[pid_int])
else
push!(union_ids, missing)
end
end
output = DataFrame(
party_id = segment_year_map.party_id,
union_party_id = union_ids,
segment_num = segment_year_map.segment_num,
country = segment_year_map.country,
year = segment_year_map.year,
rr = segment_year_map.rr
)
# Add columns for each quantity
for q in quantities
output[!, Symbol(q)] = zeros(Float64, n_rows)
output[!, Symbol("$(q)_se")] = zeros(Float64, n_rows)
output[!, Symbol("$(q)_q025")] = zeros(Float64, n_rows)
output[!, Symbol("$(q)_q975")] = zeros(Float64, n_rows)
end
println("Extracting estimates for $(n_rows) segment-year positions...")
# Progress tracking
prog_interval = max(1, n_rows ÷ 20)
for (i, row) in enumerate(eachrow(segment_year_map))
r = row.rr
# Progress
if i % prog_interval == 0 || i == n_rows
pct = round(100 * i / n_rows, digits=1)
print("\r Progress: $pct% ($i / $n_rows)")
end
for q in quantities
col_name = Symbol("$q.$r")
if !hasproperty(chains, col_name)
@warn "Column $col_name not found (rr=$r)" maxlog=5
continue
end
samples = chains[!, col_name]
# Compute summary statistics
output[i, Symbol(q)] = mean(samples)
output[i, Symbol("$(q)_se")] = std(samples)
output[i, Symbol("$(q)_q025")] = quantile(samples, 0.025)
output[i, Symbol("$(q)_q975")] = quantile(samples, 0.975)
end
end
println() # Newline after progress
# Remove the rr column from final output (internal only)
select!(output, Not(:rr))
return output
end
# =============================================================================
# STEP 5: Validation
# =============================================================================
function validate_output(output::DataFrame, segment_info::Union{DataFrame, Nothing})
println("\n" * "="^60)
println("VALIDATION CHECKS")
println("="^60)
all_passed = true
# Detect which columns are present (2D vs 4D model)
has_4d = hasproperty(output, :pro_market)
# Check 1: Range check - all estimates should be in [0, 1]
println("\n1. Range check (all values in [0, 1]):")
if has_4d
check_cols = [:pro_market, :pro_welfare, :cosmopolitan, :traditional, :economic_lr, :galtan]
else
check_cols = [:economic_lr, :galtan]
end
for col in check_cols
if !hasproperty(output, col)
continue
end
vals = output[!, col]
min_val, max_val = extrema(vals)
in_range = min_val >= 0 && max_val <= 1
status = in_range ? "PASS" : "FAIL"
println(" $col: [$(@sprintf("%.4f", min_val)), $(@sprintf("%.4f", max_val))] - $status")
all_passed = all_passed && in_range
end
# Check 2: Anchor party checks
println("\n2. Anchor party checks:")
# Define anchor parties with expected ranges (for 2D model)
# Includes both union IDs (V3) and individual constituent IDs (V4)
anchor_parties = [
(id=211, name="CDU/CSU", country="DE", econ=(0.50, 0.70), galtan=(0.45, 0.70)),
(id=1375, name="CDU", country="DE", econ=(0.50, 0.70), galtan=(0.45, 0.65)),
(id=1731, name="CSU", country="DE", econ=(0.50, 0.70), galtan=(0.55, 0.75)),
(id=383, name="SPD", country="DE", econ=(0.30, 0.50), galtan=(0.30, 0.55)),
(id=1516, name="Labour", country="GB", econ=(0.30, 0.55), galtan=(0.30, 0.55)),
(id=1567, name="Conservatives", country="GB", econ=(0.55, 0.80), galtan=(0.50, 0.75)),
(id=487, name="SAP", country="SE", econ=(0.30, 0.50), galtan=(0.35, 0.55)),
]
n_checked = 0
n_passed = 0
for anchor in anchor_parties
party_rows = filter(r -> r.party_id == anchor.id, output)
if nrow(party_rows) == 0
println(" $(anchor.name) ($(anchor.id)): NOT FOUND")
continue
end
# Use most recent 20 years of data as reference period
max_year = maximum(party_rows.year)
ref_rows = filter(r -> r.year >= max_year - 20, party_rows)
if nrow(ref_rows) == 0
ref_rows = party_rows
end
n_checked += 1
mean_econ = mean(ref_rows.economic_lr)
mean_galtan = mean(ref_rows.galtan)
econ_ok = anchor.econ[1] <= mean_econ <= anchor.econ[2]
galtan_ok = anchor.galtan[1] <= mean_galtan <= anchor.galtan[2]
all_ok = econ_ok && galtan_ok
if all_ok
n_passed += 1
end
status = all_ok ? "PASS" : "WARN"
econ_marker = econ_ok ? "" : "*"
galtan_marker = galtan_ok ? "" : "*"
@printf(" %-15s economic=%.2f%s [%.2f-%.2f] cultural=%.2f%s [%.2f-%.2f] %s\n",
anchor.name, mean_econ, econ_marker, anchor.econ[1], anchor.econ[2],
mean_galtan, galtan_marker, anchor.galtan[1], anchor.galtan[2], status)
end
if n_checked > 0
println(" Anchor check: $n_passed/$n_checked within expected ranges")
println(" Note: Model integrates text + expert data; deviations from expert-only expectations are normal")
end
# Check 3: Coverage check
println("\n3. Coverage check:")
println(" Total segment-year positions: $(nrow(output))")
println(" Unique parties: $(length(unique(output.party_id)))")
blank_country_rows = count(ismissing, output.country)
println(" Unique countries: $(length(unique(skipmissing(output.country))))")
if blank_country_rows == 0
println(" Blank country rows: 0 - PASS")
else
println(" Blank country rows: $blank_country_rows - FAIL")
all_passed = false
end
println(" Year range: $(minimum(output.year)) - $(maximum(output.year))")
# V10: Check segment distribution
segment_counts = combine(groupby(output, :party_id), nrow => :n_years,
:segment_num => (x -> length(unique(x))) => :n_segments)
parties_multi_segment = filter(:n_segments => >(1), segment_counts)
if nrow(parties_multi_segment) > 0
println("\n Parties with multiple segments: $(length(unique(parties_multi_segment.party_id)))")
end
# Check 4: No duplicates (party_id, segment_num, year should be unique)
println("\n4. Duplicate check:")
dup_count = nrow(output) - nrow(unique(select(output, :party_id, :segment_num, :year)))
if dup_count == 0
println(" No duplicate (party_id, segment_num, year) combinations - PASS")
else
println(" WARNING: Found $dup_count duplicate combinations!")
all_passed = false
end
# Check 5: SE reasonableness
println("\n5. Standard error check:")
se_cols = has_4d ?
[:pro_market_se, :pro_welfare_se, :cosmopolitan_se, :traditional_se] :
[:economic_lr_se, :galtan_se]
for col in se_cols
if !hasproperty(output, col)
continue
end
vals = output[!, col]
mean_se = mean(vals)
max_se = maximum(vals)
println(" $col: mean=$(@sprintf("%.4f", mean_se)), max=$(@sprintf("%.4f", max_se))")
end
println("\n" * "-"^60)
if all_passed
println("All validation checks PASSED")
else
println("Some validation checks FAILED - please inspect output carefully")
end
return all_passed
end
# =============================================================================
# STEP 6: Save output
# =============================================================================
function save_output(output::DataFrame, metadata::Dict, segment_info::Union{DataFrame, Nothing}, run_dir::String; outdir::String="outputs/estimations/latest")
println("\n" * "="^60)
println("SAVING OUTPUT")
println("="^60)
timestamp = Dates.format(now(), "yyyy-mm-dd_HH-MM-SS")
mkpath(outdir)
# Delete previous output files
for f in readdir(outdir)
if startswith(f, "party_positions_") && (endswith(f, ".csv") || endswith(f, ".txt") || endswith(f, ".tex"))
rm(joinpath(outdir, f))
println(" Deleted old: $f")
end
end
# Save main CSV
csv_file = joinpath(outdir, "party_positions_$timestamp.csv")
CSV.write(csv_file, output)
println("Saved: $csv_file")
println(" Rows: $(nrow(output))")
println(" Columns: $(ncol(output))")
# Count parties with multiple segments
n_multi_segment = 0
if segment_info !== nothing
party_segment_counts = combine(groupby(segment_info, :party_id), nrow => :n_segments)
n_multi_segment = count(r -> r.n_segments > 1, eachrow(party_segment_counts))
end
# Save metadata
meta_file = joinpath(outdir, "party_positions_$(timestamp)_metadata.txt")
open(meta_file, "w") do f
println(f, "Party Positions Dataset - Metadata")
println(f, "="^50)
println(f, "")
println(f, "Generated: $(Dates.format(now(), "yyyy-mm-dd HH:MM:SS"))")
println(f, "Source run: $(basename(run_dir))")
println(f, "Model file: $(get(metadata, "model_file", "unknown"))")
println(f, "")
println(f, "Dataset size:")
println(f, " Segment-year observations: $(nrow(output))")
println(f, " Unique parties: $(length(unique(output.party_id)))")
if n_multi_segment > 0
println(f, " Parties with multiple segments: $n_multi_segment")
end
println(f, " Unique countries: $(length(unique(output.country)))")
println(f, " Year range: $(minimum(output.year)) - $(maximum(output.year))")
println(f, "")
println(f, "Columns:")
println(f, " party_id: PartyFacts ID (integer) - individual party (e.g., CDU=1375, CSU=1731)")
println(f, " union_party_id: PartyFacts ID of parent union (NA for standalone parties)")
println(f, " segment_num: Segment number within party (1, 2, 3...)")
println(f, " country: ISO2 country code")
println(f, " year: Calendar year")
println(f, "")
println(f, "Segment-Based Indexing:")
println(f, " - Parties are split into segments at gaps > 7 years")
println(f, " - Each segment is estimated independently (no continuity across gaps)")
println(f, " - Segments with < 3 observations are dropped")
println(f, " - segment_num=1 is the main segment; higher numbers indicate gaps in data")
println(f, "")
# Check if this is 2D or 4D output
is_2d = !hasproperty(output, :pro_market)
if is_2d
println(f, "Model: 2D Direct Bipolar")
println(f, "")
println(f, "Bipolar scales (0 = left/cosmopolitan, 1 = right/traditionalist):")
println(f, " economic_lr: Economic left-right position (directly estimated)")
println(f, " galtan: Cultural cosmopolitan--traditionalist position (directly estimated)")
# Note: general_lr is computed internally for cross-dimensional anchoring
# but not reported as output (the two dimension-specific estimates are preferred)
else
println(f, "Model: 4D Unipolar")
println(f, "")
println(f, "Dimensions (0 = low, 1 = high):")
println(f, " pro_market: Pro-market economic position")
println(f, " pro_welfare: Pro-welfare state position")
println(f, " cosmopolitan: Cosmopolitan cultural position")
println(f, " traditional: Traditionalist cultural position")
println(f, "")
println(f, "Derived bipolar scales (0 = left/cosmopolitan, 1 = right/traditionalist):")
println(f, " economic_lr: Economic left-right (derived from pro_market - pro_welfare)")
println(f, " galtan: Cultural cosmopolitan--traditionalist (derived from traditional - cosmopolitan)")
end
println(f, "")
println(f, "Uncertainty columns:")
println(f, " *_se: Standard error (posterior SD)")
println(f, " *_q025: 2.5th percentile (lower 95% CI)")
println(f, " *_q975: 97.5th percentile (upper 95% CI)")
println(f, "")
println(f, "Model convergence:")
println(f, " Mean R-hat: $(get(metadata, "mean_rhat", "N/A"))")
println(f, " Max R-hat: $(get(metadata, "max_rhat", "N/A"))")
println(f, " Mean ESS: $(get(metadata, "mean_ess", "N/A"))")
println(f, " Min ESS: $(get(metadata, "min_ess", "N/A"))")
end
println("Saved: $meta_file")
return csv_file, meta_file
end
# =============================================================================
# STEP 5b: Verify no union/alliance IDs in output
# =============================================================================
function verify_no_unions_in_output(output::DataFrame)
println("\n" * "="^60)
println("UNION ID VERIFICATION")
println("="^60)
union_mapping_file = joinpath("data", "union_mapping.csv")
if !isfile(union_mapping_file)
println(" No union_mapping.csv found — skipping verification")
return
end
union_df = CSV.read(union_mapping_file, DataFrame)
union_pf_ids = Set(union_df.manifesto_pf_id)
output_pf_ids = Set(output.party_id)
violations = intersect(union_pf_ids, output_pf_ids)
if isempty(violations)
println(" PASS: No union/alliance PF IDs found in output")
println(" Checked $(length(union_pf_ids)) union IDs against $(length(output_pf_ids)) output parties")
else
println(" WARNING: $(length(violations)) union PF IDs found in output")
println(" (This is expected if union_mapping.csv was updated after the model run)")
for v in sort(collect(violations))
n_rows = count(r -> r.party_id == v, eachrow(output))
println(" PF $v: $n_rows rows")
end
end
end
# =============================================================================
# MAIN
# =============================================================================
function main()
println("="^60)
println("POST-ESTIMATION: Party-position model")
println("="^60)
println("Started: $(Dates.format(now(), "yyyy-mm-dd HH:MM:SS"))")
# Step 0: Find run directory (CLI --run-dir or auto-detect latest)
run_dir = nothing
output_dir = nothing
for (i, arg) in enumerate(ARGS)
if arg == "--run-dir" && i < length(ARGS)
run_dir = ARGS[i + 1]
elseif startswith(arg, "--run-dir=")
run_dir = split(arg, "=", limit=2)[2]
elseif arg == "--output-dir" && i < length(ARGS)
output_dir = ARGS[i + 1]
elseif startswith(arg, "--output-dir=")
output_dir = split(arg, "=", limit=2)[2]
end
end
if run_dir === nothing
run_dir = find_latest_run()
else
println("Using specified run directory: $run_dir")
end
# Step 1: Load run data
data = load_run_data(run_dir)
# Step 2: Build segment-year mapping
year0 = data.metadata["year0"]
segment_year_map, R, segment_info = build_segment_year_map(
data.text_data, data.expert_dim, data.expert_lr,
data.segment_info, data.segment_year_map, data.run_dir, year0
)
# Step 3: Load chains
chains = load_chains(data.chain_files)
# Step 4: Extract estimates
output = extract_estimates(chains, segment_year_map, R)
# Step 5: Validate
validate_output(output, segment_info)
# Step 5b: Verify no union/alliance IDs in output
verify_no_unions_in_output(output)
# Step 6: Save output
effective_output_dir = output_dir !== nothing ? output_dir : "outputs/estimations/latest"
csv_file, meta_file = save_output(output, data.metadata, segment_info, run_dir; outdir=effective_output_dir)
println("\n" * "="^60)
println("COMPLETE")
println("="^60)
println("Output files:")
println(" $csv_file")
println(" $meta_file")
println("\nFinished: $(Dates.format(now(), "yyyy-mm-dd HH:MM:SS"))")
return output
end
# Run if executed directly
if abspath(PROGRAM_FILE) == @__FILE__
main()
end
+332
View File
@@ -0,0 +1,332 @@
#!/usr/bin/env julia
#############################################################################
## 00_validation.jl
## Pre-flight validation checks for Stan data and initialization
## Prevents cryptic Stan errors by catching issues early
#############################################################################
using Statistics
"""
Validate Stan data dictionary before passing to Stan.
Catches common issues that cause Stan to crash with cryptic errors.
"""
function validate_stan_data(dat::Dict; verbose=true)
verbose && println("\n" * "="^70)
verbose && println("VALIDATING STAN DATA")
verbose && println("="^70)
errors = String[]
warnings = String[]
# Check for NaN/Inf in all numeric data
for (key, value) in dat
if isa(value, AbstractArray) && eltype(value) <: Number
if any(isnan, value)
push!(errors, "Data '$key' contains NaN values")
end
if any(isinf, value)
push!(errors, "Data '$key' contains Inf values")
end
elseif isa(value, Number)
if isnan(value)
push!(errors, "Data '$key' is NaN")
end
if isinf(value)
push!(errors, "Data '$key' is Inf")
end
end
end
# Validate expert data is in open interval (0, 1)
if haskey(dat, "val_dim")
val_dim = dat["val_dim"]
if any(x -> x <= 0 || x >= 1, val_dim)
n_boundary = count(x -> x <= 0 || x >= 1, val_dim)
push!(errors, "Expert dimension data has $n_boundary values at boundaries (must be in (0,1))")
if verbose
println(" Dimension-specific expert data range: [$(minimum(val_dim)), $(maximum(val_dim))]")
end
end
end
if haskey(dat, "val_lr")
val_lr = dat["val_lr"]
if any(x -> x <= 0 || x >= 1, val_lr)
n_boundary = count(x -> x <= 0 || x >= 1, val_lr)
push!(errors, "Expert L-R data has $n_boundary values at boundaries (must be in (0,1))")
if verbose
println(" L-R expert data range: [$(minimum(val_lr)), $(maximum(val_lr))]")
end
end
end
# Validate manifesto data
if haskey(dat, "positive") && haskey(dat, "sample")
positive = dat["positive"]
sample = dat["sample"]
if any(positive .> sample)
push!(errors, "Manifesto: positive counts exceed sample sizes")
end
if any(positive .< 0)
push!(errors, "Manifesto: negative positive counts found")
end
if any(sample .< 0)
push!(errors, "Manifesto: negative sample sizes found")
end
end
# Validate indices are within bounds
# V10: Check segment indices (ss_man) if present, otherwise party indices (jj_man)
if haskey(dat, "S") && haskey(dat, "ss_man")
S = dat["S"]
ss_man = dat["ss_man"]
if any(ss_man .< 1) || any(ss_man .> S)
push!(errors, "Manifesto segment indices out of bounds [1, $S]")
end
elseif haskey(dat, "J") && haskey(dat, "jj_man")
J = dat["J"]
jj_man = dat["jj_man"]
if any(jj_man .< 1) || any(jj_man .> J)
push!(errors, "Manifesto party indices out of bounds [1, $J]")
end
end
if haskey(dat, "R") && haskey(dat, "rr_man")
R = dat["R"]
rr_man = dat["rr_man"]
if any(rr_man .< 1) || any(rr_man .> R)
push!(errors, "Manifesto party-year indices out of bounds [1, $R]")
end
end
# V4: Validate constituent arrays
if haskey(dat, "const_rr_man") && haskey(dat, "R")
R = dat["R"]
const_rr = dat["const_rr_man"]
if any(const_rr .< 1) || any(const_rr .> R)
push!(errors, "const_rr_man out of bounds [1, $R]")
end
# Verify offsets are valid
if haskey(dat, "const_offset_man") && haskey(dat, "n_const_man")
offsets = dat["const_offset_man"]
n_consts = dat["n_const_man"]
total = dat["N_const_man_total"]
for i in eachindex(offsets)
if offsets[i] + n_consts[i] - 1 > total
push!(errors, "const_offset_man[$i] + n_const_man[$i] exceeds N_const_man_total")
break
end
end
end
end
# Print summary
if verbose
println("\nDATA SUMMARY:")
# V10: Show segments if present
if haskey(dat, "S")
println(" Segments (S): $(dat["S"])")
println(" Parties with segments (J): $(get(dat, "J", "N/A"))")
else
println(" Parties (J): $(get(dat, "J", "N/A"))")
end
println(" Countries (P): $(get(dat, "P", "N/A"))")
println(" Segment-years (R): $(get(dat, "R", "N/A"))")
println(" Years (T_year): $(get(dat, "T_year", "N/A"))")
println(" Manifesto obs: $(get(dat, "N_man", "N/A"))")
println(" Expert dim obs: $(get(dat, "N_exp_dim", "N/A"))")
println(" Expert L-R obs: $(get(dat, "N_exp_lr", "N/A"))")
if haskey(dat, "mn_resp_log_man")
println("\nPRIOR MEANS:")
println(" Manifesto: $(round(dat["mn_resp_log_man"], digits=3))")
println(" Expert dim: $(round(dat["mn_resp_log_exp_dim"], digits=3))")
println(" Expert L-R: $(round(dat["mn_resp_log_exp_lr"], digits=3))")
end
end
# Report results
if !isempty(errors)
println("\n❌ VALIDATION FAILED - $(length(errors)) ERROR(S):")
for (i, err) in enumerate(errors)
println(" $i. $err")
end
return false
end
if !isempty(warnings)
println("\n⚠️ $(length(warnings)) WARNING(S):")
for (i, warn) in enumerate(warnings)
println(" $i. $warn")
end
end
if verbose
println("\n✓ DATA VALIDATION PASSED")
println("="^70)
end
return true
end
"""
Validate initialization values before passing to Stan.
Checks for common issues that cause immediate Stan crashes.
"""
function validate_init_values(init_dict::Dict; verbose=true)
verbose && println("\n" * "="^70)
verbose && println("VALIDATING INITIALIZATION VALUES")
verbose && println("="^70)
errors = String[]
warnings = String[]
for (key, value) in init_dict
# Check for NaN/Inf
if isa(value, AbstractArray) && eltype(value) <: Number
if any(isnan, value)
push!(errors, "Init '$key' contains NaN")
end
if any(isinf, value)
push!(errors, "Init '$key' contains Inf")
end
if verbose && length(value) > 0
val_array = vec(value)
println(" $key: range [$(round(minimum(val_array), digits=3)), $(round(maximum(val_array), digits=3))]")
end
elseif isa(value, Number)
if isnan(value)
push!(errors, "Init '$key' is NaN")
end
if isinf(value)
push!(errors, "Init '$key' is Inf")
end
if verbose
println(" $key: $(round(value, digits=3))")
end
end
# Check positive constraints (common Stan constraints)
# Exception: *_raw parameters are non-centered and can be any real
if (contains(string(key), "sigma") || contains(string(key), "tau") || contains(string(key), "phi")) &&
!endswith(string(key), "_raw")
if isa(value, Number) && value <= 0
push!(errors, "Init '$key' = $value violates constraint > 0")
elseif isa(value, AbstractArray) && any(value .<= 0)
push!(errors, "Init '$key' has values ≤ 0 (violates constraint > 0)")
end
end
# Check Cholesky factors are valid
if contains(string(key), "L_Omega")
if isa(value, AbstractMatrix)
# Check it's lower triangular with positive diagonal
n = size(value, 1)
if size(value, 2) != n
push!(errors, "Init '$key' is not square")
end
for i in 1:n
if value[i, i] <= 0
push!(errors, "Init '$key' has non-positive diagonal at position $i")
end
for j in (i+1):n
if abs(value[i, j]) > 1e-10
push!(warnings, "Init '$key' is not lower triangular")
break
end
end
end
end
end
# Check slope parameters for positive constraint (V2/V3 feature)
if key == "Gamma_man_slope_raw"
if isa(value, AbstractArray) && any(value .< 0)
push!(errors, "Init 'Gamma_man_slope_raw' has negative values (must be ≥ 0)")
end
end
end
# Report results
if !isempty(errors)
println("\n❌ INIT VALIDATION FAILED - $(length(errors)) ERROR(S):")
for (i, err) in enumerate(errors)
println(" $i. $err")
end
return false
end
if !isempty(warnings)
println("\n⚠️ $(length(warnings)) WARNING(S):")
for (i, warn) in enumerate(warnings)
println(" $i. $warn")
end
end
if verbose
println("\n✓ INIT VALIDATION PASSED")
println("="^70)
end
return true
end
"""
Estimate memory requirements for model
"""
function estimate_memory_requirements(dat::Dict; verbose=true, num_chains::Int=4, num_samples::Int=1000, num_threads_per_chain::Int=1)
if !verbose
return
end
println("\n" * "="^70)
println("MEMORY ESTIMATE")
println("="^70)
R = get(dat, "R", 0)
J = get(dat, "J", 0)
K_man = get(dat, "K_man", 0)
K_exp_dim = get(dat, "K_exp_dim", 0)
K_exp_lr = get(dat, "K_exp_lr", 0)
N_man = get(dat, "N_man", 0)
N_ciy = get(dat, "N_ciy", 0)
T_year = get(dat, "T_year", 0)
# Rough parameter count
theta_params = 4 * R
item_params = 4 * K_man * 2 + K_exp_dim * 3 + K_exp_lr * 2
strategic_params = get(dat, "P", 0) * get(dat, "K_man", 0) # Country-item intercepts
other_params = 4 * J + T_year + J + N_ciy + 50
total_params = theta_params + item_params + strategic_params + other_params
# Memory estimate (very rough)
# Each parameter: ~8 bytes (float64) × samples × chains
total_draws_per_param = num_samples * num_chains
bytes_per_param = 8 * total_draws_per_param
total_mb = (total_params * bytes_per_param) / (1024 * 1024)
println(" Configuration:")
println(" Chains: $num_chains")
println(" Samples per chain: $num_samples")
println(" Threads per chain: $num_threads_per_chain")
println(" Total parallel workers: $(num_chains * num_threads_per_chain)")
println(" Total parameters: ~$(total_params)")
println(" Estimated memory (samples only): ~$(round(total_mb, digits=0)) MB")
thread_scaling = max(1, num_threads_per_chain)
println(" With thread overhead (×$(thread_scaling)): ~$(round(total_mb * thread_scaling, digits=0)) MB")
println(" With safety margin (×3): ~$(round(3 * total_mb * thread_scaling, digits=0)) MB")
if total_mb * thread_scaling * 3 > 8000
println("\n⚠️ WARNING: Model may require > 8GB RAM")
end
println("="^70)
end
+277
View File
@@ -0,0 +1,277 @@
#!/usr/bin/env julia
#############################################################################
## 02_data_loading.jl
## Load and preprocess data for latent trait model
## Loads three datasets: text_data (manifesto + PolDem), expert dimension-specific, expert general L-R
##
## Supports both:
## - 4D model (V10): type_high/type_low columns for bipolar bridges
## - 2D model (V1): dim_idx + direction columns for direct estimation
#############################################################################
using DataFrames, CSV, CategoricalArrays, Statistics
#############################################################################
## UNION MAPPING: Individual party estimates via mean-constituent model
## Loads data/union_mapping.csv and builds lookup structures
#############################################################################
"""
load_union_mapping(project_root::String)
Load union_mapping.csv and build lookup dictionaries.
Returns (union_to_constituents, constituent_to_union) dicts.
If file is missing or empty, returns empty dicts (backwards compatible).
"""
function load_union_mapping(project_root::String=".")
mapping_file = joinpath(project_root, "data", "union_mapping.csv")
union_to_constituents = Dict{Int, Vector{Int}}()
constituent_to_union = Dict{Int, Int}()
if !isfile(mapping_file)
println(" No union_mapping.csv found - running without union decomposition")
return union_to_constituents, constituent_to_union
end
df = CSV.read(mapping_file, DataFrame)
if nrow(df) == 0
println(" union_mapping.csv is empty - running without union decomposition")
return union_to_constituents, constituent_to_union
end
for row in eachrow(df)
union_id = row.manifesto_pf_id
expert_id = row.expert_pf_id
if !haskey(union_to_constituents, union_id)
union_to_constituents[union_id] = Int[]
end
if !(expert_id in union_to_constituents[union_id])
push!(union_to_constituents[union_id], expert_id)
end
constituent_to_union[expert_id] = union_id
end
println(" Union mapping loaded: $(length(union_to_constituents)) unions, $(length(constituent_to_union)) constituents")
return union_to_constituents, constituent_to_union
end
#############################################################################
## SEGMENT-BASED INDEXING CONFIGURATION
## Split parties at gaps > MAX_GAP years to avoid flat posteriors
#############################################################################
const MAX_GAP = 7 # Maximum years between observations within a segment
const MIN_OBS = 2 # Minimum observations per segment (drop segments with fewer)
#############################################################################
## 2D MODEL MAPPING CONFIGURATION
## Maps type_high/type_low pairs to dim_idx + direction
#############################################################################
const TYPE_TO_DIM_DIRECTION = Dict(
# Economic dimension: pro_market = right (+1), pro_welfare = left (-1)
("pro_market", "pro_welfare") => (dim_idx=1, direction=1), # Right
("pro_welfare", "pro_market") => (dim_idx=1, direction=-1), # Left
# Cultural dimension: traditional = TAN (+1), cosmopolitan = GAL (-1)
("traditional", "cosmopolitan") => (dim_idx=2, direction=1), # TAN
("cosmopolitan", "traditional") => (dim_idx=2, direction=-1) # GAL
)
# Expert dimension mapping (lrecon -> economic, galtan/cultural -> galtan)
const EXPERT_VAR_TO_DIM = Dict(
"lrecon_ches" => 1,
"lrecon_vparty" => 1,
"welf_vparty" => 1,
"lrecon_gps" => 1,
"lrecon_poppa" => 1,
"galtan_ches" => 2,
"libcon_gps" => 2,
"immig_vparty" => 2,
"lgbt_vparty" => 2,
"culsup_vparty" => 2,
"relig_vparty" => 2,
"gender_vparty" => 2
)
function load_and_preprocess_4dim_data(start_year=1950; data_dir::String=".")
println("Loading party-position data files...")
println("Start year filter: $start_year")
data_dir != "." && println("Data directory: $data_dir")
# Load union mapping (check data_dir first, fall back to project root)
println("\nLoading union mapping...")
union_mapping_dir = isfile(joinpath(data_dir, "data", "union_mapping.csv")) ? data_dir : "."
union_to_constituents, constituent_to_union = load_union_mapping(union_mapping_dir)
# Load the three datasets
text_data_raw = CSV.read(joinpath(data_dir, "text_data.csv"), DataFrame)
expert_raw = CSV.read(joinpath(data_dir, "expert.csv"), DataFrame)
lr_data_raw = CSV.read(joinpath(data_dir, "lr_data.csv"), DataFrame)
# Filter to start year BEFORE calculating year0
text_data_raw = text_data_raw[text_data_raw.year .>= start_year, :]
expert_raw = expert_raw[expert_raw.year .>= start_year, :]
lr_data_raw = lr_data_raw[lr_data_raw.year .>= start_year, :]
println("Data filtered to $start_year onwards:")
println(" Text data (manifesto + PolDem): $(nrow(text_data_raw)) observations")
println(" Expert: $(nrow(expert_raw)) observations")
println(" L-R data: $(nrow(lr_data_raw)) observations")
# Define base year for relative time indexing
year0 = Int(minimum([minimum(text_data_raw.year), minimum(expert_raw.year), minimum(lr_data_raw.year)])) - 1
println("Base year set to: $year0")
# Create type mapping for the four dimensions
type_map = Dict(
"pro_market" => 1,
"pro_welfare" => 2,
"cosmopolitan" => 3,
"traditional" => 4
)
println("Type mapping: pro_market=1, pro_welfare=2, cosmopolitan=3, traditional=4")
# Process text data (manifesto + PolDem media)
text_data = copy(text_data_raw)
text_data = text_data[text_data.year .> year0, :]
# Add type indices for text items (V4/V10: bipolar bridge structure)
if !("type_high" in names(text_data)) || !("type_low" in names(text_data))
error("Text data must contain 'type_high' and 'type_low' columns with values: pro_market, pro_welfare, cosmopolitan, traditional")
end
text_data.type_high_idx = [type_map[t] for t in text_data.type_high]
text_data.type_low_idx = [type_map[t] for t in text_data.type_low]
# V1 (2D model): Add dim_idx and direction columns
# Maps type_high/type_low to single dimension + direction
dim_idx_man = Int[]
direction_man = Int[]
for row in eachrow(text_data)
key = (row.type_high, row.type_low)
if haskey(TYPE_TO_DIM_DIRECTION, key)
mapping = TYPE_TO_DIM_DIRECTION[key]
push!(dim_idx_man, mapping.dim_idx)
push!(direction_man, mapping.direction)
else
# Unknown mapping - this should not happen with valid data
error("Unknown type_high/type_low pair: $(row.type_high) / $(row.type_low)")
end
end
text_data.dim_idx_man = dim_idx_man
text_data.direction_man = direction_man
# Standard processing
text_data.country = categorical(text_data.country)
text_data.party = categorical(text_data.party)
text_data.var = categorical(text_data.var)
text_data.Year = Int.(text_data.year) .- year0
sort!(text_data, [:country, :party, :year, :var])
println("Text data processed: $(nrow(text_data)) observations with bipolar bridge structure")
# Process expert dimension-specific data (bipolar items like lrecon_ches, galtan_ches)
expert_dim_vars = ["lrecon_ches", "galtan_ches", "lrecon_vparty", "welf_vparty",
"lrecon_gps", "libcon_gps", "lrecon_poppa",
"immig_vparty", "lgbt_vparty", "culsup_vparty", "relig_vparty", "gender_vparty"]
expert_dim = expert_raw[in.(expert_raw.var, Ref(expert_dim_vars)), :]
expert_dim = expert_dim[(expert_dim.year .> year0) .& (expert_dim.val .>= 0) .& (expert_dim.val .<= 1), :]
# V5: Load integer observations, scale sizes, and expert counts for beta-binomial likelihood
expert_dim.val_int = Int.(expert_dim.val_int)
expert_dim.n_scale = Int.(expert_dim.n_scale)
expert_dim.n_experts = Int.(expert_dim.n_experts)
# Add type mappings for dimension-specific expert data
if !("type_low" in names(expert_dim)) || !("type_high" in names(expert_dim))
error("Expert data must contain 'type_low' and 'type_high' columns")
end
expert_dim.type_high_idx = [type_map[t] for t in expert_dim.type_high]
expert_dim.type_low_idx = [type_map[t] for t in expert_dim.type_low]
# V1 (2D model): Add dim_idx for expert dimension data
dim_idx_exp = Int[]
for row in eachrow(expert_dim)
var_name = string(row.var)
if haskey(EXPERT_VAR_TO_DIM, var_name)
push!(dim_idx_exp, EXPERT_VAR_TO_DIM[var_name])
else
# Fallback: infer from type_high/type_low
key = (row.type_high, row.type_low)
if haskey(TYPE_TO_DIM_DIRECTION, key)
push!(dim_idx_exp, TYPE_TO_DIM_DIRECTION[key].dim_idx)
else
error("Unknown expert variable: $var_name with type pair $(row.type_high) / $(row.type_low)")
end
end
end
expert_dim.dim_idx_exp = dim_idx_exp
expert_dim.country = categorical(expert_dim.country)
expert_dim.party = categorical(expert_dim.party)
expert_dim.var = categorical(expert_dim.var)
expert_dim.Year = Int.(expert_dim.year) .- year0
sort!(expert_dim, [:country, :party, :year, :var])
println("Expert dimension-specific data processed: $(nrow(expert_dim)) observations")
# Process expert general L-R data (cross-dimensional anchoring)
lr_vars = ["lr_ches", "lr_poppa", "lr_morgan"] # General left-right items
expert_lr = lr_data_raw[in.(lr_data_raw.var, Ref(lr_vars)), :]
expert_lr = expert_lr[(expert_lr.year .> year0) .& (expert_lr.val .>= 0) .& (expert_lr.val .<= 1), :]
# V5: Load integer observations, scale sizes, and expert counts for beta-binomial likelihood
expert_lr.val_int = Int.(expert_lr.val_int)
expert_lr.n_scale = Int.(expert_lr.n_scale)
expert_lr.n_experts = Int.(expert_lr.n_experts)
expert_lr.country = categorical(expert_lr.country)
expert_lr.party = categorical(expert_lr.party)
expert_lr.var = categorical(expert_lr.var)
expert_lr.Year = Int.(expert_lr.year) .- year0
sort!(expert_lr, [:country, :party, :year, :var])
println("Expert general L-R data processed: $(nrow(expert_lr)) observations")
# Validate data integrity
println("\nData validation:")
# Check text data dimension pair distribution (V4: bipolar bridges)
type_pair_counts = combine(groupby(text_data, [:type_high, :type_low]), nrow => :count)
for row in eachrow(type_pair_counts)
println(" $(row.type_high) ↔ $(row.type_low): $(row.count) text data observations")
end
# Check expert dimension-specific type pairs
type_pair_counts = combine(groupby(expert_dim, [:type_high, :type_low]), nrow => :count)
for row in eachrow(type_pair_counts)
println(" $(row.type_high) - $(row.type_low): $(row.count) expert dimension-specific observations")
end
# Check general L-R items
lr_var_counts = combine(groupby(expert_lr, :var), nrow => :count)
for row in eachrow(lr_var_counts)
println(" $(row.var): $(row.count) general L-R observations")
end
# Check overlapping parties across datasets
text_data_parties = Set(text_data.party)
expert_dim_parties = Set(expert_dim.party)
expert_lr_parties = Set(expert_lr.party)
all_parties = union(text_data_parties, expert_dim_parties, expert_lr_parties)
println("\nParty coverage:")
println(" Total unique parties: $(length(all_parties))")
println(" In text data: $(length(text_data_parties))")
println(" In expert dimension-specific: $(length(expert_dim_parties))")
println(" In expert general L-R: $(length(expert_lr_parties))")
println(" In all three datasets: $(length(intersect(text_data_parties, expert_dim_parties, expert_lr_parties)))")
return text_data, expert_dim, expert_lr, year0, union_to_constituents, constituent_to_union
end
# Execute if run directly
if abspath(PROGRAM_FILE) == @__FILE__
text_data, expert_dim, expert_lr, year0, u2c, c2u = load_and_preprocess_4dim_data()
println("4D data loading test completed successfully")
end
+947
View File
@@ -0,0 +1,947 @@
#!/usr/bin/env julia
#############################################################################
## 03_data_preparation_4dim.jl
## V10: Segment-based indexing to fix long gap interpolation issues
##
## Key change: Split parties at gaps > MAX_GAP years into independent segments.
## Each segment has its own random walk (restarts at segment start).
## Segments with < MIN_OBS observations are dropped.
#############################################################################
using DataFrames, CSV, CategoricalArrays, Statistics, StatsFuns
# Import configuration from data loading module
include("02_data_loading.jl")
"""
split_party_years_into_segments(years::Vector{Int}, max_gap::Int)
Split a party's observation years into segments based on gaps.
Returns a vector of vectors, where each inner vector contains consecutive years
with max `max_gap` years between observations.
"""
function split_party_years_into_segments(years::Vector{Int}, max_gap::Int)
if isempty(years)
return Vector{Vector{Int}}()
end
sorted_years = sort(unique(years))
segments = [Int[sorted_years[1]]]
for y in sorted_years[2:end]
if y - segments[end][end] <= max_gap
push!(segments[end], y)
else
# Gap too large - start new segment
push!(segments, [y])
end
end
return segments
end
function prepare_4dim_stan_data(manifesto, expert_dim, expert_lr, year0;
union_to_constituents=Dict{Int,Vector{Int}}(),
constituent_to_union=Dict{Int,Int}())
println("Preparing data for Stan model (Segment-based indexing)...")
println(" MAX_GAP = $MAX_GAP years, MIN_OBS = $MIN_OBS observations")
has_unions = !isempty(union_to_constituents)
if has_unions
println(" Union mapping: $(length(union_to_constituents)) unions, $(length(constituent_to_union)) constituents")
else
println(" No union mapping - standard party indexing")
end
# =========================================================================
# STEP 1: Collect observation years per party (union-aware)
# For union parties: create segments for each CONSTITUENT, not the union.
# Union manifesto years are assigned to all constituents.
# =========================================================================
# Identify which party IDs in data are unions vs standalone
# NOTE: levels() returns raw types (Int64 for integer party IDs).
# We consistently use String keys for all party lookups to avoid type mismatches.
manifesto_party_strs = Set(string.(levels(manifesto.party)))
union_ids_in_data = Set{String}()
if has_unions
for uid in keys(union_to_constituents)
uid_str = string(uid)
if uid_str in manifesto_party_strs
push!(union_ids_in_data, uid_str)
end
end
println(" Union party IDs found in manifesto data: $(length(union_ids_in_data))")
end
# Collect all party IDs that need segments
# For unions: constituents get segments; union itself does NOT
# For standalone: party gets segment as before
party_obs_years = Dict{String, Set{Int}}()
# First pass: collect non-union parties from all data sources (as strings)
all_data_parties = Set{String}()
for p in levels(manifesto.party)
push!(all_data_parties, string(p))
end
for p in levels(expert_dim.party)
push!(all_data_parties, string(p))
end
for p in levels(expert_lr.party)
push!(all_data_parties, string(p))
end
# Initialize observation years for standalone parties and constituents
for p in all_data_parties
p_int = tryparse(Int, string(p))
if p_int !== nothing && has_unions && haskey(union_to_constituents, p_int) && string(p) in union_ids_in_data
# This is a union ID in manifesto - skip it, create entries for constituents instead
continue
end
party_obs_years[p] = Set{Int}()
end
# For unions: ensure all constituents have entries
if has_unions
for (uid, constituents) in union_to_constituents
uid_str = string(uid)
if uid_str in union_ids_in_data
for cid in constituents
cid_str = string(cid)
if !haskey(party_obs_years, cid_str)
party_obs_years[cid_str] = Set{Int}()
end
end
end
end
end
# Add years from manifesto
for row in eachrow(manifesto)
p_str = string(row.party)
p_int = tryparse(Int, p_str)
if p_int !== nothing && has_unions && haskey(union_to_constituents, p_int) && p_str in union_ids_in_data
# Union manifesto: add year to ALL constituents
for cid in union_to_constituents[p_int]
cid_str = string(cid)
if haskey(party_obs_years, cid_str)
push!(party_obs_years[cid_str], row.Year)
end
end
else
if haskey(party_obs_years, p_str)
push!(party_obs_years[p_str], row.Year)
end
end
end
# Add years from expert_dim (individual party data - direct)
for row in eachrow(expert_dim)
p_str = string(row.party)
if haskey(party_obs_years, p_str)
push!(party_obs_years[p_str], row.Year)
end
end
# Add years from expert_lr (individual party data - direct)
for row in eachrow(expert_lr)
p_str = string(row.party)
if haskey(party_obs_years, p_str)
push!(party_obs_years[p_str], row.Year)
end
end
J_original = length(party_obs_years)
println("Total number of parties to create segments for: $J_original")
# =========================================================================
# STEP 2: Split parties into segments and filter by MIN_OBS
# =========================================================================
println("\nCreating segments (splitting at gaps > $MAX_GAP years)...")
segment_data = []
segment_id = 0
parties_split = 0
segments_dropped = 0
observations_dropped = 0
for (p, years_set) in party_obs_years
years = collect(years_set)
if isempty(years)
continue
end
segments = split_party_years_into_segments(years, MAX_GAP)
if length(segments) > 1
parties_split += 1
end
for (seg_num, seg_years) in enumerate(segments)
n_obs = length(seg_years)
if n_obs >= MIN_OBS
segment_id += 1
push!(segment_data, (
segment_id = segment_id,
party_id = p,
segment_num = seg_num,
year_start = minimum(seg_years),
year_end = maximum(seg_years),
n_obs = n_obs
))
else
segments_dropped += 1
observations_dropped += n_obs
end
end
end
segment_info = DataFrame(segment_data)
S = nrow(segment_info) # Number of valid segments
println(" Segments created: $S (from $J_original parties)")
println(" Parties split into multiple segments: $parties_split")
println(" Segments dropped (< $MIN_OBS obs): $segments_dropped")
println(" Observations dropped: $observations_dropped")
# Get unique parties that have at least one valid segment
all_parties = unique(segment_info.party_id)
J = length(all_parties)
println(" Parties with valid segments: $J")
# Create party-to-index mapping for the valid parties
party_to_index = Dict(all_parties .=> 1:J)
# =========================================================================
# STEP 3: Create segment-year index space (R) - consecutive within segment
# =========================================================================
println("\nCreating segment-year index space...")
segment_year_data = []
for row in eachrow(segment_info)
for y in row.year_start:row.year_end
push!(segment_year_data, (
segment_id = row.segment_id,
party_id = row.party_id,
Year = y
))
end
end
segment_year = DataFrame(segment_year_data)
segment_year.rr = 1:nrow(segment_year)
R = nrow(segment_year)
println("Total segment-year positions (R): $R")
# Compute len_theta_ts for segments (years per segment)
len_theta_ts = [row.year_end - row.year_start + 1 for row in eachrow(segment_info)]
@assert sum(len_theta_ts) == R "sum(len_theta_ts)=$(sum(len_theta_ts)) must equal R=$R"
# =========================================================================
# STEP 4: Map observations to segment-year indices (union-aware)
# =========================================================================
println("\nMapping observations to segment-year indices...")
# Create lookup: (party_str, year) -> segment_id (for valid segments only)
party_year_to_segment = Dict{Tuple{String, Int}, Int}()
for row in eachrow(segment_info)
for y in row.year_start:row.year_end
party_year_to_segment[(string(row.party_id), y)] = row.segment_id
end
end
# --- MANIFESTO: union-aware mapping ---
# For union manifesto obs: map to first constituent's segment (for ss_man).
# The actual theta averaging is handled via constituent arrays.
manifesto_segment_ids = Union{Int, Missing}[]
for row in eachrow(manifesto)
p_str = string(row.party)
p_int = tryparse(Int, p_str)
key = (p_str, row.Year)
if haskey(party_year_to_segment, key)
# Direct mapping (non-union or constituent with own segment)
push!(manifesto_segment_ids, party_year_to_segment[key])
elseif p_int !== nothing && has_unions && haskey(union_to_constituents, p_int)
# Union party: use first constituent's segment
found = false
for cid in union_to_constituents[p_int]
ckey = (string(cid), row.Year)
if haskey(party_year_to_segment, ckey)
push!(manifesto_segment_ids, party_year_to_segment[ckey])
found = true
break
end
end
if !found
push!(manifesto_segment_ids, missing)
end
else
push!(manifesto_segment_ids, missing)
end
end
manifesto.segment_id = manifesto_segment_ids
n_manifesto_before = nrow(manifesto)
manifesto = manifesto[.!ismissing.(manifesto.segment_id), :]
manifesto.segment_id = Int.(manifesto.segment_id)
println(" Manifesto: $(nrow(manifesto))/$n_manifesto_before observations (dropped $(n_manifesto_before - nrow(manifesto)) in invalid segments)")
# --- EXPERT DIM: direct mapping (individual party data) ---
expert_dim_segment_ids = Union{Int, Missing}[]
for row in eachrow(expert_dim)
key = (string(row.party), row.Year)
if haskey(party_year_to_segment, key)
push!(expert_dim_segment_ids, party_year_to_segment[key])
else
push!(expert_dim_segment_ids, missing)
end
end
expert_dim.segment_id = expert_dim_segment_ids
n_expert_dim_before = nrow(expert_dim)
expert_dim = expert_dim[.!ismissing.(expert_dim.segment_id), :]
expert_dim.segment_id = Int.(expert_dim.segment_id)
println(" Expert dim: $(nrow(expert_dim))/$n_expert_dim_before observations (dropped $(n_expert_dim_before - nrow(expert_dim)) in invalid segments)")
# --- EXPERT LR: direct mapping (individual party data) ---
expert_lr_segment_ids = Union{Int, Missing}[]
for row in eachrow(expert_lr)
key = (string(row.party), row.Year)
if haskey(party_year_to_segment, key)
push!(expert_lr_segment_ids, party_year_to_segment[key])
else
push!(expert_lr_segment_ids, missing)
end
end
expert_lr.segment_id = expert_lr_segment_ids
n_expert_lr_before = nrow(expert_lr)
expert_lr = expert_lr[.!ismissing.(expert_lr.segment_id), :]
expert_lr.segment_id = Int.(expert_lr.segment_id)
println(" Expert LR: $(nrow(expert_lr))/$n_expert_lr_before observations (dropped $(n_expert_lr_before - nrow(expert_lr)) in invalid segments)")
# =========================================================================
# STEP 5: Create segment indices (ss) for each observation
# ss indexes into 1:S (segment space), used for party-level parameters
# =========================================================================
# Create segment_id to ss mapping
segment_to_ss = Dict(row.segment_id => i for (i, row) in enumerate(eachrow(segment_info)))
manifesto.ss_man = [segment_to_ss[sid] for sid in manifesto.segment_id]
expert_dim.ss_exp_dim = [segment_to_ss[sid] for sid in expert_dim.segment_id]
expert_lr.ss_exp_lr = [segment_to_ss[sid] for sid in expert_lr.segment_id]
# Validate segment indices
@assert all(1 .<= manifesto.ss_man .<= S)
@assert all(1 .<= expert_dim.ss_exp_dim .<= S)
@assert all(1 .<= expert_lr.ss_exp_lr .<= S)
# =========================================================================
# STEP 6: Map rr indices (segment-year) to datasets via leftjoin
# =========================================================================
# Create (segment_id, Year) -> rr lookup
seg_year_to_rr = Dict{Tuple{Int, Int}, Int}()
for row in eachrow(segment_year)
seg_year_to_rr[(row.segment_id, row.Year)] = row.rr
end
# Join manifesto with segment_year to get rr indices
manifesto = leftjoin(manifesto, segment_year, on=[:segment_id, :Year])
rename!(manifesto, :rr => :rr_man)
expert_dim = leftjoin(expert_dim, segment_year, on=[:segment_id, :Year])
rename!(expert_dim, :rr => :rr_exp_dim)
expert_lr = leftjoin(expert_lr, segment_year, on=[:segment_id, :Year])
rename!(expert_lr, :rr => :rr_exp_lr)
# Validate rr mappings
@assert all(!ismissing, manifesto.rr_man) "Some manifesto observations have no rr_man mapping"
@assert all(!ismissing, expert_dim.rr_exp_dim) "Some expert_dim observations have no rr_exp_dim mapping"
@assert all(!ismissing, expert_lr.rr_exp_lr) "Some expert_lr observations have no rr_exp_lr mapping"
# Convert to Int
manifesto.rr_man = Int.(manifesto.rr_man)
expert_dim.rr_exp_dim = Int.(expert_dim.rr_exp_dim)
expert_lr.rr_exp_lr = Int.(expert_lr.rr_exp_lr)
# Validate bounds
@assert all(1 .<= manifesto.rr_man .<= R) "rr_man out of bounds"
@assert all(1 .<= expert_dim.rr_exp_dim .<= R) "rr_exp_dim out of bounds"
@assert all(1 .<= expert_lr.rr_exp_lr .<= R) "rr_exp_lr out of bounds"
# Print diagnostics
n_observed = length(unique(vcat(manifesto.rr_man, expert_dim.rr_exp_dim, expert_lr.rr_exp_lr)))
println("\nSegment-years with data: $n_observed / $R ($(round(100*n_observed/R, digits=1))%)")
println("Segment-years for interpolation: $(R - n_observed)")
# =========================================================================
# STEP 6b: Build constituent arrays for union manifesto/expert observations
# For each manifesto obs: store list of constituent rr indices for averaging
# Non-union obs: single rr (n_const=1)
# Union obs: multiple rr values (n_const=len(constituents))
# =========================================================================
println("\nBuilding constituent arrays for mean-constituent model...")
# --- Manifesto constituent arrays ---
n_const_man_vec = Int[] # n_const for each manifesto obs
const_rr_man_vec = Int[] # flat array of constituent rr values
const_offset_man_vec = Int[] # offset into const_rr for each obs
offset = 1
for row in eachrow(manifesto)
p_str = string(row.party)
p_int = tryparse(Int, p_str)
if p_int !== nothing && has_unions && haskey(union_to_constituents, p_int) && p_str in union_ids_in_data
# Union manifesto obs: find rr for each constituent in this year
constituent_rrs = Int[]
for cid in union_to_constituents[p_int]
ckey = (string(cid), row.Year)
if haskey(party_year_to_segment, ckey)
sid = party_year_to_segment[ckey]
rr_key = (sid, row.Year)
if haskey(seg_year_to_rr, rr_key)
push!(constituent_rrs, seg_year_to_rr[rr_key])
end
end
end
if isempty(constituent_rrs)
# Fallback: use the rr_man already assigned
push!(constituent_rrs, row.rr_man)
end
push!(n_const_man_vec, length(constituent_rrs))
push!(const_offset_man_vec, offset)
append!(const_rr_man_vec, constituent_rrs)
offset += length(constituent_rrs)
else
# Non-union: single constituent (itself)
push!(n_const_man_vec, 1)
push!(const_offset_man_vec, offset)
push!(const_rr_man_vec, row.rr_man)
offset += 1
end
end
N_const_man_total = length(const_rr_man_vec)
n_union_man = count(x -> x > 1, n_const_man_vec)
println(" Manifesto: $(nrow(manifesto)) obs, $n_union_man union obs, $N_const_man_total total constituent entries")
# --- Expert dim constituent arrays ---
# Individual expert obs always have n_const=1 (direct mapping)
# Union-level expert obs (if any) would average — but typically expert data is at individual party level
n_const_exp_dim_vec = Int[]
const_rr_exp_dim_vec = Int[]
const_offset_exp_dim_vec = Int[]
offset = 1
for row in eachrow(expert_dim)
p_str = string(row.party)
p_int = tryparse(Int, p_str)
if p_int !== nothing && has_unions && haskey(union_to_constituents, p_int) && p_str in union_ids_in_data
# Union-level expert obs: average over constituents
constituent_rrs = Int[]
for cid in union_to_constituents[p_int]
ckey = (string(cid), row.Year)
if haskey(party_year_to_segment, ckey)
sid = party_year_to_segment[ckey]
rr_key = (sid, row.Year)
if haskey(seg_year_to_rr, rr_key)
push!(constituent_rrs, seg_year_to_rr[rr_key])
end
end
end
if isempty(constituent_rrs)
push!(constituent_rrs, row.rr_exp_dim)
end
push!(n_const_exp_dim_vec, length(constituent_rrs))
push!(const_offset_exp_dim_vec, offset)
append!(const_rr_exp_dim_vec, constituent_rrs)
offset += length(constituent_rrs)
else
# Individual party obs
push!(n_const_exp_dim_vec, 1)
push!(const_offset_exp_dim_vec, offset)
push!(const_rr_exp_dim_vec, row.rr_exp_dim)
offset += 1
end
end
N_const_exp_dim_total = length(const_rr_exp_dim_vec)
n_union_exp_dim = count(x -> x > 1, n_const_exp_dim_vec)
println(" Expert dim: $(nrow(expert_dim)) obs, $n_union_exp_dim union obs, $N_const_exp_dim_total total constituent entries")
# --- Expert LR constituent arrays ---
n_const_exp_lr_vec = Int[]
const_rr_exp_lr_vec = Int[]
const_offset_exp_lr_vec = Int[]
offset = 1
for row in eachrow(expert_lr)
p_str = string(row.party)
p_int = tryparse(Int, p_str)
if p_int !== nothing && has_unions && haskey(union_to_constituents, p_int) && p_str in union_ids_in_data
constituent_rrs = Int[]
for cid in union_to_constituents[p_int]
ckey = (string(cid), row.Year)
if haskey(party_year_to_segment, ckey)
sid = party_year_to_segment[ckey]
rr_key = (sid, row.Year)
if haskey(seg_year_to_rr, rr_key)
push!(constituent_rrs, seg_year_to_rr[rr_key])
end
end
end
if isempty(constituent_rrs)
push!(constituent_rrs, row.rr_exp_lr)
end
push!(n_const_exp_lr_vec, length(constituent_rrs))
push!(const_offset_exp_lr_vec, offset)
append!(const_rr_exp_lr_vec, constituent_rrs)
offset += length(constituent_rrs)
else
push!(n_const_exp_lr_vec, 1)
push!(const_offset_exp_lr_vec, offset)
push!(const_rr_exp_lr_vec, row.rr_exp_lr)
offset += 1
end
end
N_const_exp_lr_total = length(const_rr_exp_lr_vec)
n_union_exp_lr = count(x -> x > 1, n_const_exp_lr_vec)
println(" Expert LR: $(nrow(expert_lr)) obs, $n_union_exp_lr union obs, $N_const_exp_lr_total total constituent entries")
# =========================================================================
# STEP 7: Filter manifesto items with sufficient observations
# =========================================================================
var_man_counts_df = combine(groupby(manifesto, :var), :var => length => :n_obs)
var_man_counts_df = var_man_counts_df[var_man_counts_df.n_obs .>= 2, :]
var_man_counts = var_man_counts_df.var
var_exp_dim_counts_df = combine(groupby(expert_dim, :var), :var => length => :n_obs)
var_exp_dim_counts_df = var_exp_dim_counts_df[var_exp_dim_counts_df.n_obs .>= 2, :]
var_exp_dim_counts = var_exp_dim_counts_df.var
var_exp_lr_counts_df = combine(groupby(expert_lr, :var), :var => length => :n_obs)
var_exp_lr_counts_df = var_exp_lr_counts_df[var_exp_lr_counts_df.n_obs .>= 2, :]
var_exp_lr_counts = var_exp_lr_counts_df.var
manifesto = manifesto[in.(manifesto.var, Ref(var_man_counts)), :]
manifesto.var_man = levelcode.(categorical(manifesto.var, levels=unique(var_man_counts)))
expert_dim = expert_dim[in.(expert_dim.var, Ref(var_exp_dim_counts)), :]
expert_dim.var_exp_dim = levelcode.(categorical(expert_dim.var, levels=unique(var_exp_dim_counts)))
expert_lr = expert_lr[in.(expert_lr.var, Ref(var_exp_lr_counts)), :]
expert_lr.var_exp_lr = levelcode.(categorical(expert_lr.var, levels=unique(var_exp_lr_counts)))
# =========================================================================
# STEP 8: Country (group) indexing
# =========================================================================
all_groups = unique(vcat(levels(manifesto.country), levels(expert_dim.country), levels(expert_lr.country)))
P = length(all_groups)
println("Total number of unique countries (P): $P")
group_to_index = Dict(all_groups .=> 1:P)
manifesto.pp_man = [group_to_index[c] for c in manifesto.country]
expert_dim.pp_exp_dim = [group_to_index[c] for c in expert_dim.country]
expert_lr.pp_exp_lr = [group_to_index[c] for c in expert_lr.country]
@assert all(1 .<= manifesto.pp_man .<= P)
@assert all(1 .<= expert_dim.pp_exp_dim .<= P)
@assert all(1 .<= expert_lr.pp_exp_lr .<= P)
# =========================================================================
# STEP 9: Country-item-year combinations for zero-inflation model
# =========================================================================
manifesto.ciy_key = string.(manifesto.country, "_", manifesto.var_man, "_", manifesto.Year)
ciy_keys = unique(manifesto.ciy_key)
N_ciy = length(ciy_keys)
println("Total unique country-item-year combinations (N_ciy): $N_ciy")
ciy_key_to_index = Dict(ciy_keys .=> 1:N_ciy)
manifesto.ciy_idx = [ciy_key_to_index[k] for k in manifesto.ciy_key]
@assert all(1 .<= manifesto.ciy_idx .<= N_ciy)
# =========================================================================
# STEP 10: Segment-country mapping (each segment inherits from party)
# For constituents: inherit country from their union's manifesto data
# =========================================================================
party_country_dict = Dict{String, Int}()
# From data directly
for row in eachrow(manifesto)
p_str = string(row.party)
p_int = tryparse(Int, p_str)
c_idx = group_to_index[string(row.country)]
party_country_dict[p_str] = c_idx
# If union, also assign country to all constituents
if p_int !== nothing && has_unions && haskey(union_to_constituents, p_int)
for cid in union_to_constituents[p_int]
party_country_dict[string(cid)] = c_idx
end
end
end
for row in eachrow(expert_dim)
party_country_dict[string(row.party)] = group_to_index[string(row.country)]
end
for row in eachrow(expert_lr)
party_country_dict[string(row.party)] = group_to_index[string(row.country)]
end
# Map each segment to its party's country
segment_country_idx = Int[]
for row in eachrow(segment_info)
pid = string(row.party_id)
if haskey(party_country_dict, pid)
push!(segment_country_idx, party_country_dict[pid])
else
# Fallback: try to find via union mapping
pf_int = tryparse(Int, pid)
if pf_int !== nothing && haskey(constituent_to_union, pf_int)
uid = constituent_to_union[pf_int]
if haskey(party_country_dict, string(uid))
push!(segment_country_idx, party_country_dict[string(uid)])
else
error("Cannot find country for constituent $pid (union $uid)")
end
else
error("Cannot find country for party $pid")
end
end
end
@assert all(1 .<= segment_country_idx .<= P)
@assert length(segment_country_idx) == S
# Persist canonical country on segment mapping tables
segment_country = [all_groups[idx] for idx in segment_country_idx]
segment_info.country = segment_country
segment_country_by_id = Dict(row.segment_id => segment_country[i] for (i, row) in enumerate(eachrow(segment_info)))
segment_year.country = [segment_country_by_id[sid] for sid in segment_year.segment_id]
# =========================================================================
# STEP 11: Load party family data and map to segments
# =========================================================================
println("\nLoading party family data...")
party_families_file = joinpath("data", "party_families.csv")
if !isfile(party_families_file)
error("party_families.csv not found at $party_families_file")
end
party_families_df = CSV.read(party_families_file, DataFrame)
pf_to_family = Dict(row.partyfacts_id => row.family for row in eachrow(party_families_df))
all_family_names = unique(party_families_df.family)
family_to_idx = Dict(f => i for (i, f) in enumerate(all_family_names))
F = length(all_family_names)
println("Total number of party families (F): $F")
println("Family categories: ", join(all_family_names, ", "))
# Map each segment to its party's family
# For constituents: try own family first, fall back to union's family
segment_family_idx = Int[]
unmatched_segments = String[]
default_family_idx = haskey(family_to_idx, "other") ? family_to_idx["other"] : 1
for row in eachrow(segment_info)
pf_id = tryparse(Int, string(row.party_id))
if !isnothing(pf_id) && haskey(pf_to_family, pf_id)
family_name = pf_to_family[pf_id]
push!(segment_family_idx, family_to_idx[family_name])
elseif !isnothing(pf_id) && haskey(constituent_to_union, pf_id) && haskey(pf_to_family, constituent_to_union[pf_id])
# Fall back to union's family
family_name = pf_to_family[constituent_to_union[pf_id]]
push!(segment_family_idx, family_to_idx[family_name])
else
push!(segment_family_idx, default_family_idx)
push!(unmatched_segments, "$(row.party_id)_seg$(row.segment_num)")
end
end
if !isempty(unmatched_segments)
println(" Warning: $(length(unmatched_segments)) segments not matched to families (assigned to 'other')")
if length(unmatched_segments) <= 10
println(" Unmatched: ", join(unmatched_segments, ", "))
end
end
@assert all(1 .<= segment_family_idx .<= F)
@assert length(segment_family_idx) == S
# =========================================================================
# STEP 12: Find anchor segment
# With unions: use CDU (1375) as anchor (individual constituent)
# Without unions: use CDU/CSU (211) as anchor (union ID)
# =========================================================================
anchor_party_id = has_unions ? 1375 : 211
anchor_label = has_unions ? "CDU" : "CDU/CSU"
anchor_segments = filter(row -> tryparse(Int, string(row.party_id)) == anchor_party_id, segment_info)
if nrow(anchor_segments) > 0
# Pick segment with most observations
anchor_segment_idx = anchor_segments[argmax(anchor_segments.n_obs), :segment_id]
# Convert to ss index (1:S)
anchor_segment_ss = segment_to_ss[anchor_segment_idx]
println(" Anchor segment ($anchor_label, ID $anchor_party_id): segment $anchor_segment_ss ($(anchor_segments[argmax(anchor_segments.n_obs), :n_obs]) obs)")
else
println(" Warning: Anchor party $anchor_label (ID $anchor_party_id) not found, using segment 1")
anchor_segment_ss = 1
end
# =========================================================================
# STEP 13: Validation - check no long gaps remain within segments
# =========================================================================
println("\nValidating segment structure...")
rr_to_segment_year = Dict(row.rr => (segment_id=row.segment_id, year=row.Year) for row in eachrow(segment_year))
seg_obs_years = Dict(row.segment_id => Set{Int}() for row in eachrow(segment_info))
for row in eachrow(manifesto)
push!(seg_obs_years[row.segment_id], row.Year)
end
for row in eachrow(expert_dim)
push!(seg_obs_years[row.segment_id], row.Year)
end
for row in eachrow(expert_lr)
push!(seg_obs_years[row.segment_id], row.Year)
end
# Union manifesto rows contribute to every constituent through const_rr_man_vec,
# even though row.segment_id stores only a representative segment for indexing.
# Include these constituent rr values so validation matches the actual Stan data.
for rr in const_rr_man_vec
if haskey(rr_to_segment_year, rr)
sy = rr_to_segment_year[rr]
push!(seg_obs_years[sy.segment_id], sy.year)
end
end
max_internal_gap = 0
for (i, row) in enumerate(eachrow(segment_info))
seg_obs = collect(seg_obs_years[row.segment_id])
if length(seg_obs) > 1
gaps = diff(sort(unique(seg_obs)))
if !isempty(gaps)
max_gap_in_seg = maximum(gaps)
max_internal_gap = max(max_internal_gap, max_gap_in_seg)
if max_gap_in_seg > MAX_GAP
@warn "Segment $i (party $(row.party_id)) has internal gap of $max_gap_in_seg years!"
end
end
end
end
println(" Maximum internal gap within segments: $max_internal_gap years (limit: $MAX_GAP)")
# Validate all segments have >= MIN_OBS
for row in eachrow(segment_info)
@assert row.n_obs >= MIN_OBS "Segment $(row.segment_id) has only $(row.n_obs) obs (min: $MIN_OBS)"
end
println(" All segments have >= $MIN_OBS observations: PASS")
println("\nSegment-based indexing created successfully")
println(" S (segments): $S")
println(" R (segment-years): $R")
println(" J (parties with valid segments): $J")
return (manifesto=manifesto, expert_dim=expert_dim, expert_lr=expert_lr,
segment_year=segment_year, segment_info=segment_info,
all_parties=all_parties, all_groups=all_groups,
S=S, J=J, P=P, R=R, N_ciy=N_ciy, len_theta_ts=len_theta_ts,
segment_country_idx=segment_country_idx, group_to_index=group_to_index,
F=F, segment_family_idx=segment_family_idx, anchor_segment_idx=anchor_segment_ss,
# Constituent arrays for mean-constituent model
N_const_man_total=N_const_man_total,
n_const_man=n_const_man_vec,
const_offset_man=const_offset_man_vec,
const_rr_man=const_rr_man_vec,
N_const_exp_dim_total=N_const_exp_dim_total,
n_const_exp_dim=n_const_exp_dim_vec,
const_offset_exp_dim=const_offset_exp_dim_vec,
const_rr_exp_dim=const_rr_exp_dim_vec,
N_const_exp_lr_total=N_const_exp_lr_total,
n_const_exp_lr=n_const_exp_lr_vec,
const_offset_exp_lr=const_offset_exp_lr_vec,
const_rr_exp_lr=const_rr_exp_lr_vec,
union_to_constituents=union_to_constituents,
constituent_to_union=constituent_to_union)
end
function finalize_4dim_stan_data(manifesto, expert_dim, expert_lr, segment_year, segment_info,
all_parties, all_groups, group_to_index, year0, S, J, P, R, N_ciy,
len_theta_ts, segment_country_idx, F, segment_family_idx, anchor_segment_idx;
N_const_man_total=0, n_const_man=Int[], const_offset_man=Int[], const_rr_man=Int[],
N_const_exp_dim_total=0, n_const_exp_dim=Int[], const_offset_exp_dim=Int[], const_rr_exp_dim=Int[],
N_const_exp_lr_total=0, n_const_exp_lr=Int[], const_offset_exp_lr=Int[], const_rr_exp_lr=Int[])
println("Finalizing 4D Stan data structure (V10: segment-based)...")
# Map years for temporal indexing
years_all = sort(unique(vcat(manifesto.Year, expert_dim.Year, expert_lr.Year)))
T_year = length(years_all)
year_map = DataFrame(year_rel=years_all, year_ix=1:T_year)
# Apply year mapping to all datasets
manifesto = leftjoin(manifesto, year_map, on=[:Year => :year_rel])
rename!(manifesto, :year_ix => :year_for_man)
expert_dim = leftjoin(expert_dim, year_map, on=[:Year => :year_rel])
rename!(expert_dim, :year_ix => :year_for_exp_dim)
expert_lr = leftjoin(expert_lr, year_map, on=[:Year => :year_rel])
rename!(expert_lr, :year_ix => :year_for_exp_lr)
# Validate year assignments
@assert all(.!ismissing.(manifesto.year_for_man))
@assert all(.!ismissing.(expert_dim.year_for_exp_dim))
@assert all(.!ismissing.(expert_lr.year_for_exp_lr))
@assert all(1 .<= manifesto.year_for_man .<= T_year)
@assert all(1 .<= expert_dim.year_for_exp_dim .<= T_year)
@assert all(1 .<= expert_lr.year_for_exp_lr .<= T_year)
# Add small epsilon to prevent exact zeros and ones
epsilon = 1e-6
expert_dim.val = clamp.(expert_dim.val, epsilon, 1.0 - epsilon)
expert_lr.val = clamp.(expert_lr.val, epsilon, 1.0 - epsilon)
# Calculate prior means
man_positive_sample = manifesto.positive[manifesto.sample .> 0] ./ manifesto.sample[manifesto.sample .> 0]
mn_resp_log_man = StatsFuns.logit(mean(man_positive_sample))
mn_resp_log_exp_dim = StatsFuns.logit(mean(expert_dim.val))
mn_resp_log_exp_lr = StatsFuns.logit(mean(expert_lr.val))
println("Prior means calculated:")
println(" Manifesto: $(round(mn_resp_log_man, digits=3))")
println(" Expert dimension-specific: $(round(mn_resp_log_exp_dim, digits=3))")
println(" Expert general L-R: $(round(mn_resp_log_exp_lr, digits=3))")
# V6: Decade indexing for hierarchical L-R weights
expert_lr_actual_years = expert_lr.Year .+ year0
expert_lr_decade_raw = div.(expert_lr_actual_years, 10)
all_lr_decades = sort(unique(expert_lr_decade_raw))
lr_decade_to_index = Dict(all_lr_decades .=> 1:length(all_lr_decades))
dd_exp_lr = [lr_decade_to_index[d] for d in expert_lr_decade_raw]
D_lr = length(all_lr_decades)
println(" Decade indexing (V6): $D_lr decades, range $(minimum(all_lr_decades)*10)s-$(maximum(all_lr_decades)*10)s")
# Create Stan data dictionary - V10 uses S (segments) instead of J (parties)
dat_4dim = Dict(
# Common data - V10: S = number of segments
"S" => S, # NEW: Number of segments (was J)
"J" => J, # Keep J for reference (parties with valid segments)
"P" => P,
"R" => R,
"T_year" => T_year,
"len_theta_ts" => Int.(len_theta_ts),
# Segment-country mapping (V10: segments inherit country from party)
"segment_country" => segment_country_idx,
# Segment family data (V10: segments inherit family from party)
"F" => F,
"segment_family" => segment_family_idx,
# Anchor segment for identification (CDU/CSU segment)
"anchor_segment" => anchor_segment_idx,
# Manifesto data - use ss_man (segment index) instead of jj_man (party index)
"N_man" => nrow(manifesto),
"K_man" => length(unique(manifesto.var_man)),
"kk_man" => manifesto.var_man,
"ss_man" => manifesto.ss_man, # V10: segment index (was jj_man)
"rr_man" => manifesto.rr_man,
"pp_man" => manifesto.pp_man,
"positive" => manifesto.positive,
"sample" => manifesto.sample,
"year_for_man" => manifesto.year_for_man,
"type_high_idx_man" => manifesto.type_high_idx,
"type_low_idx_man" => manifesto.type_low_idx,
# V1 (2D model): dimension index and direction for text data
"dim_idx_man" => manifesto.dim_idx_man,
"direction_man" => manifesto.direction_man,
# Country-item-year data for zero-inflation
"N_ciy" => N_ciy,
"ciy_idx" => manifesto.ciy_idx,
# Expert dimension-specific data
"N_exp_dim" => nrow(expert_dim),
"K_exp_dim" => length(unique(expert_dim.var_exp_dim)),
"kk_exp_dim" => expert_dim.var_exp_dim,
"ss_exp_dim" => expert_dim.ss_exp_dim, # V10: segment index
"rr_exp_dim" => expert_dim.rr_exp_dim,
"pp_exp_dim" => expert_dim.pp_exp_dim,
# V5 K-scaling: use rounded sum (mean × K × n_scale) and total trials (K × n_scale)
"val_dim_int" => Int.(clamp.(round.(expert_dim.val .* expert_dim.n_scale .* expert_dim.n_experts), 0, expert_dim.n_scale .* expert_dim.n_experts)),
"n_total_exp_dim" => expert_dim.n_scale .* expert_dim.n_experts,
"n_experts_exp_dim" => expert_dim.n_experts,
"type_high_idx" => expert_dim.type_high_idx,
"type_low_idx" => expert_dim.type_low_idx,
# V1 (2D model): dimension index for expert dimension data
"dim_idx_exp" => expert_dim.dim_idx_exp,
# Expert general L-R data
"N_exp_lr" => nrow(expert_lr),
"K_exp_lr" => length(unique(expert_lr.var_exp_lr)),
"kk_exp_lr" => expert_lr.var_exp_lr,
"ss_exp_lr" => expert_lr.ss_exp_lr, # V10: segment index
"rr_exp_lr" => expert_lr.rr_exp_lr,
"pp_exp_lr" => expert_lr.pp_exp_lr,
# V5 K-scaling: use rounded sum (mean × K × n_scale) and total trials (K × n_scale)
"val_lr_int" => Int.(clamp.(round.(expert_lr.val .* expert_lr.n_scale .* expert_lr.n_experts), 0, expert_lr.n_scale .* expert_lr.n_experts)),
"n_total_exp_lr" => expert_lr.n_scale .* expert_lr.n_experts,
"n_experts_exp_lr" => expert_lr.n_experts,
# V6: Decade indexing for hierarchical L-R weights
"D_lr" => D_lr,
"dd_exp_lr" => dd_exp_lr,
# Prior information
"mn_resp_log_man" => mn_resp_log_man,
"mn_resp_log_exp_dim" => mn_resp_log_exp_dim,
"mn_resp_log_exp_lr" => mn_resp_log_exp_lr,
# Constituent arrays for mean-constituent model (V4)
"N_const_man_total" => max(1, N_const_man_total),
"n_const_man" => isempty(n_const_man) ? ones(Int, nrow(manifesto)) : n_const_man,
"const_offset_man" => isempty(const_offset_man) ? collect(1:nrow(manifesto)) : const_offset_man,
"const_rr_man" => isempty(const_rr_man) ? manifesto.rr_man : const_rr_man,
"N_const_exp_dim_total" => max(1, N_const_exp_dim_total),
"n_const_exp_dim" => isempty(n_const_exp_dim) ? ones(Int, nrow(expert_dim)) : n_const_exp_dim,
"const_offset_exp_dim" => isempty(const_offset_exp_dim) ? collect(1:nrow(expert_dim)) : const_offset_exp_dim,
"const_rr_exp_dim" => isempty(const_rr_exp_dim) ? expert_dim.rr_exp_dim : const_rr_exp_dim,
"N_const_exp_lr_total" => max(1, N_const_exp_lr_total),
"n_const_exp_lr" => isempty(n_const_exp_lr) ? ones(Int, nrow(expert_lr)) : n_const_exp_lr,
"const_offset_exp_lr" => isempty(const_offset_exp_lr) ? collect(1:nrow(expert_lr)) : const_offset_exp_lr,
"const_rr_exp_lr" => isempty(const_rr_exp_lr) ? expert_lr.rr_exp_lr : const_rr_exp_lr
)
println("4D Stan data dictionary created with $(length(dat_4dim)) elements")
# Print summary statistics
println("\nData summary (V10: Segment-based):")
println(" Segments: $(dat_4dim["S"])")
println(" Parties with valid segments: $(dat_4dim["J"])")
println(" Segment-year combinations: $(dat_4dim["R"])")
println(" Manifesto observations: $(dat_4dim["N_man"])")
println(" Expert dimension-specific observations: $(dat_4dim["N_exp_dim"])")
println(" Expert general L-R observations: $(dat_4dim["N_exp_lr"])")
println(" Unique manifesto items: $(dat_4dim["K_man"])")
println(" Unique expert dimension-specific items: $(dat_4dim["K_exp_dim"])")
println(" Unique expert general L-R items: $(dat_4dim["K_exp_lr"])")
println(" Years: $(dat_4dim["T_year"])")
return (dat_4dim=dat_4dim, manifesto=manifesto, expert_dim=expert_dim,
expert_lr=expert_lr, T_year=T_year, segment_year=segment_year,
segment_info=segment_info)
end
# Execute if run directly
if abspath(PROGRAM_FILE) == @__FILE__
println("Run from main script to execute the full 4D pipeline")
end
File diff suppressed because it is too large Load Diff
+174
View File
@@ -0,0 +1,174 @@
#!/usr/bin/env julia
#############################################################################
## 05_results_processing.jl
## Extract and process 4D model results with diagnostics
## Extract and process model results without election effects
#############################################################################
using StanSample, DataFrames, Statistics
function extract_model_results_4dim(stanmodel)
"""
Extract model results for the party-position model
Simplified version - no election effects (pure latent traits)
"""
println("Extracting 4D model results...")
try
println("Model completed successfully - extracting results")
# Save the full stanmodel object for downstream processing
# Post-estimation will extract specific parameters later
return (
samples = stanmodel, # Save the full stanmodel with all MCMC samples
extraction_status = "success"
)
catch e
println("Error in result extraction: $e")
return (
samples = "error",
extraction_status = "error",
error_message = string(e)
)
end
end
function compute_model_diagnostics(stanmodel_result)
"""
Compute convergence diagnostics from Stan model
Returns R-hat, ESS statistics, and overall convergence assessment
stanmodel_result can be either:
- A SampleModel object directly
- A named tuple from run_4dim_stan_model containing .stanmodel
"""
println("Computing model diagnostics...")
try
# Handle both direct SampleModel and named tuple from run_4dim_stan_model
stanmodel = if hasproperty(stanmodel_result, :stanmodel)
stanmodel_result.stanmodel
else
stanmodel_result
end
# Get REAL diagnostics using StanSample.jl
diagnostics_summary = read_summary(stanmodel)
# Extract real Rhat and ESS values. Stan summary column names differ
# across CmdStan/StanSample versions, so resolve aliases explicitly.
summary_names = names(diagnostics_summary)
rhat_col = if "r_hat" in summary_names
"r_hat"
elseif "R_hat" in summary_names
"R_hat"
elseif "RHat" in summary_names
"RHat"
else
error("No R-hat column found in summary. Columns: $(join(summary_names, ", "))")
end
ess_col = if "ess_bulk" in summary_names
"ess_bulk"
elseif "ess" in summary_names
"ess"
elseif "ESS_bulk" in summary_names
"ESS_bulk"
elseif "n_eff" in summary_names
"n_eff"
else
error("No ESS column found in summary. Columns: $(join(summary_names, ", "))")
end
rhat_vals = diagnostics_summary[!, rhat_col]
ess_bulk_vals = diagnostics_summary[!, ess_col]
# Compute real statistics (handle NaN values properly)
# Use isfinite to exclude both missing and NaN values
valid_rhat = filter(isfinite, rhat_vals)
valid_ess = filter(isfinite, ess_bulk_vals)
mean_rhat = length(valid_rhat) > 0 ? mean(valid_rhat) : NaN
max_rhat = length(valid_rhat) > 0 ? maximum(valid_rhat) : NaN
mean_ess = length(valid_ess) > 0 ? mean(valid_ess) : NaN
min_ess = length(valid_ess) > 0 ? minimum(valid_ess) : NaN
# Count problematic parameters (use isfinite for consistent counting)
high_rhat_count = count(x -> isfinite(x) && x > 1.1, rhat_vals)
moderate_rhat_count = count(x -> isfinite(x) && x > 1.05, rhat_vals)
low_ess_count = count(x -> isfinite(x) && x < 400, ess_bulk_vals)
very_low_ess_count = count(x -> isfinite(x) && x < 100, ess_bulk_vals)
# Total parameter count
total_params = length(valid_rhat)
# Overall assessment (handle NaN values)
if isnan(max_rhat) || total_params == 0
convergence_status = "insufficient_data"
else
excellent_convergence = max_rhat < 1.05 && high_rhat_count == 0 && very_low_ess_count == 0
good_convergence = max_rhat < 1.1 && high_rhat_count < 5 && very_low_ess_count < total_params * 0.1
acceptable_convergence = max_rhat < 1.2 && high_rhat_count < total_params * 0.1
if excellent_convergence
convergence_status = "excellent"
elseif good_convergence
convergence_status = "good"
elseif acceptable_convergence
convergence_status = "acceptable"
else
convergence_status = "poor"
end
end
println("\nDiagnostics computed:")
println(" Total parameters: $total_params")
println(" Mean R-hat: $(round(mean_rhat, digits=4))")
println(" Max R-hat: $(round(max_rhat, digits=4))")
println(" High R-hat count (>1.1): $high_rhat_count")
println(" Mean ESS: $(round(mean_ess, digits=0))")
println(" Min ESS: $(round(min_ess, digits=0))")
println(" Very low ESS count (<100): $very_low_ess_count")
println(" Convergence status: $convergence_status")
return (
diagnostics_summary = diagnostics_summary,
convergence_status = convergence_status,
mean_rhat = mean_rhat,
max_rhat = max_rhat,
mean_ess = mean_ess,
min_ess = min_ess,
high_rhat_count = high_rhat_count,
moderate_rhat_count = moderate_rhat_count,
low_ess_count = low_ess_count,
very_low_ess_count = very_low_ess_count,
total_params = total_params
)
catch e
println("Error in diagnostics computation: $e")
println("Stack trace:")
showerror(stdout, e, catch_backtrace())
return (
diagnostics_summary = "error",
convergence_status = "error",
mean_rhat = 999.0,
max_rhat = 999.0,
mean_ess = 0.0,
min_ess = 0.0,
high_rhat_count = 999,
moderate_rhat_count = 999,
low_ess_count = 999,
very_low_ess_count = 999,
total_params = 0,
error_message = string(e)
)
end
end
# Execute if run directly
if abspath(PROGRAM_FILE) == @__FILE__
println("Run from main run_model.jl to execute the full pipeline")
end
+428
View File
@@ -0,0 +1,428 @@
#!/usr/bin/env julia
#############################################################################
## 06_save_model.jl
## CSV-First Save Architecture
## Robust, portable, no serialization issues
#############################################################################
module RobustSave
using Dates, Printf, CSV, DataFrames, JSON
export robust_save_model_csv, robust_save_model
"""
CSV-First Save Architecture
When chains_already_saved=true (BULLETPROOF MODE):
- Chains already saved to run_dir/chains/ by model execution
- Just verify chains and add metadata/data files
- Even if this function crashes, chains are SAFE
When chains_already_saved=false (legacy mode):
- Copy chains from temp directory to new run directory
- Add metadata/data files
Saves model results as:
1. CSV chain files (source of truth - never fail)
2. Data CSVs (original inputs for reproducibility)
3. Simple metadata.json (no complex types)
4. Human-readable README.txt
No JLD2, no serialization issues, fully portable and reproducible.
"""
function robust_save_model_csv(
run_dir_or_temp::String,
data_dict::Dict,
original_data::Dict,
metadata::Dict;
chains_already_saved::Bool=false
)
println("\n" * "=" ^ 70)
if chains_already_saved
println("ADDING METADATA TO EXISTING RUN (chains already secured)")
else
println("CSV-FIRST MODEL SAVE")
end
println("=" ^ 70)
# Determine directories based on mode
if chains_already_saved
# Chains already saved - run_dir_or_temp IS the run directory
run_dir = run_dir_or_temp
chains_dir = joinpath(run_dir, "chains")
data_dir = joinpath(run_dir, "data")
run_id = basename(run_dir)
timestamp = replace(run_id, "run_" => "")
println("Run directory: $run_dir")
println("Mode: Chains already secured, adding metadata")
# Verify chains directory exists
if !isdir(chains_dir)
error("CRITICAL: Chains directory not found: $chains_dir")
end
# Count existing chain files
chain_files = filter(f -> endswith(f, ".csv") && contains(f, "chain"), readdir(chains_dir))
if isempty(chain_files)
error("CRITICAL: No chain CSV files found in $chains_dir")
end
println("Found $(length(chain_files)) chain files already saved")
# Calculate total size
total_size_gb = 0.0
for chain_file in chain_files
chain_path = joinpath(chains_dir, chain_file)
total_size_gb += filesize(chain_path) / (1024^3)
end
println("✓ Chains verified ($(round(total_size_gb, digits=2)) GB total)")
else
# Legacy mode - create new run directory and copy chains
temp_csv_dir = run_dir_or_temp
timestamp = Dates.format(Dates.now(), "yyyy-mm-dd_HH-MM-SS")
run_id = "run_$(timestamp)"
run_dir = joinpath("outputs", "model_outputs", "latest", run_id)
chains_dir = joinpath(run_dir, "chains")
data_dir = joinpath(run_dir, "data")
println("Run ID: $run_id")
println("Output directory: $run_dir")
# Create directory structure
mkpath(chains_dir)
# STEP 1: Copy CSV chain files
println("\n" * "=" ^ 70)
println("STEP 1: Copying MCMC chain CSV files")
println("=" ^ 70)
println("Source: $temp_csv_dir")
println("Destination: $chains_dir")
csv_files = filter(f -> endswith(f, ".csv"), readdir(temp_csv_dir))
chain_files = filter(f -> contains(f, "chain"), csv_files)
if isempty(chain_files)
error("No chain CSV files found in $temp_csv_dir")
end
println("Found $(length(chain_files)) chain files")
total_size_gb = 0.0
for (i, csv_file) in enumerate(sort(chain_files))
src_path = joinpath(temp_csv_dir, csv_file)
# Rename to standard format: chain_1.csv, chain_2.csv, etc.
dest_filename = "chain_$i.csv"
dest_path = joinpath(chains_dir, dest_filename)
src_size = filesize(src_path)
size_gb = src_size / (1024^3)
total_size_gb += size_gb
println(" Copying $csv_file → $dest_filename ($(round(size_gb, digits=2)) GB)")
cp(src_path, dest_path, force=true)
# Verify copy with size check
dest_size = filesize(dest_path)
if dest_size != src_size
error("CRITICAL: Size mismatch for $dest_filename! Source: $src_size, Dest: $dest_size")
end
end
println("✓ All chains copied and verified ($(round(total_size_gb, digits=2)) GB total)")
end
# Create data directory
mkpath(data_dir)
# STEP 2: Save data CSVs
println("\n" * "=" ^ 70)
println("STEP 2: Saving original data CSVs")
println("=" ^ 70)
for (name, df) in original_data
if isa(df, DataFrame)
csv_path = joinpath(data_dir, "$(name).csv")
println(" Saving $(name).csv ($(nrow(df)) rows)")
CSV.write(csv_path, df)
end
end
println("✓ Data CSVs saved")
# STEP 3: Save Stan data dictionary as JSON
println("\n" * "=" ^ 70)
println("STEP 3: Saving Stan data dictionary")
println("=" ^ 70)
# Convert data_dict to JSON-serializable format
stan_data_json = Dict{String, Any}()
for (k, v) in data_dict
try
# Only save simple types (numbers, arrays of numbers)
if isa(v, Number) || isa(v, AbstractArray{<:Number})
stan_data_json[k] = v
elseif isa(v, AbstractArray)
# Try to convert, skip if fails
try
stan_data_json[k] = collect(v)
catch
println(" Skipping $k (complex type)")
end
end
catch e
println(" Warning: Could not serialize $k: $e")
end
end
stan_data_path = joinpath(data_dir, "stan_data.json")
open(stan_data_path, "w") do f
JSON.print(f, stan_data_json, 2)
end
println("✓ Stan data saved to stan_data.json")
# Count chain files for metadata
chain_files_final = filter(f -> endswith(f, ".csv") && contains(f, "chain"), readdir(chains_dir))
num_chains = length(chain_files_final)
# Recalculate total_size_gb if in chains_already_saved mode
if chains_already_saved
total_size_gb = 0.0
for chain_file in chain_files_final
chain_path = joinpath(chains_dir, chain_file)
total_size_gb += filesize(chain_path) / (1024^3)
end
end
# STEP 4: Save metadata
println("\n" * "=" ^ 70)
println("STEP 4: Saving metadata")
println("=" ^ 70)
# Add run info to metadata
metadata["run_id"] = run_id
metadata["timestamp"] = timestamp
metadata["files"] = Dict(
"chains" => ["chains/chain_$i.csv" for i in 1:num_chains],
"data" => readdir(data_dir),
"chain_size_gb" => num_chains > 0 ? round(total_size_gb / num_chains, digits=2) : 0.0,
"total_size_gb" => round(total_size_gb, digits=2)
)
metadata_path = joinpath(run_dir, "metadata.json")
open(metadata_path, "w") do f
JSON.print(f, metadata, 2)
end
println("✓ Metadata saved to metadata.json")
# STEP 5: Generate README
println("\n" * "=" ^ 70)
println("STEP 5: Generating README")
println("=" ^ 70)
readme_path = joinpath(run_dir, "README.txt")
generate_readme(readme_path, run_id, metadata, num_chains, total_size_gb)
println("✓ README generated")
# STEP 6: Final verification
println("\n" * "=" ^ 70)
println("STEP 6: Verification")
println("=" ^ 70)
# Verify all chain files exist and are readable
all_good = true
verified_files = Dict{String, Dict{String, Any}}()
for i in 1:num_chains
chain_path = joinpath(chains_dir, "chain_$i.csv")
if !isfile(chain_path)
println(" ✗ Missing: chain_$i.csv")
all_good = false
else
# Quick read test and size check
try
CSV.File(chain_path; limit=1)
file_size_gb = filesize(chain_path) / (1024^3)
verified_files["chain_$i.csv"] = Dict(
"path" => chain_path,
"size_gb" => file_size_gb,
"verified" => true
)
println(" ✓ chain_$i.csv verified ($(round(file_size_gb, digits=2)) GB)")
catch e
println(" ✗ Cannot read chain_$i.csv: $e")
all_good = false
end
end
end
if !all_good
error("Verification failed - some files are missing or corrupted")
end
println("\n" * "=" ^ 70)
println("✓ MODEL SAVED SUCCESSFULLY")
println("=" ^ 70)
println("Run directory: $run_dir")
println("Total size: $(round(total_size_gb, digits=2)) GB")
println("Status: All files verified and ready")
println("=" ^ 70)
# Return verification details for cleanup
return (
run_dir = run_dir,
verified_files = verified_files,
total_size_gb = total_size_gb,
verification_passed = all_good,
num_chains = num_chains
)
end
function generate_readme(
filepath::String,
run_id::String,
metadata::Dict,
num_chains::Int,
total_size_gb::Float64
)
"""Generate human-readable README file"""
open(filepath, "w") do f
write(f, "=" ^ 78 * "\n")
write(f, "PARTY-POSITION MODEL - MODEL RUN RESULTS\n")
write(f, "=" ^ 78 * "\n\n")
write(f, "Run ID: $run_id\n")
write(f, "Model: $(get(metadata, "model_file", "unknown"))\n")
write(f, "Date: $(Dates.format(Dates.now(), "yyyy-mm-dd HH:MM:SS"))\n")
write(f, "Status: $(get(metadata, "convergence_status", "unknown"))\n\n")
write(f, "=" ^ 78 * "\n")
write(f, "DIRECTORY CONTENTS\n")
write(f, "=" ^ 78 * "\n\n")
write(f, "chains/\n")
for i in 1:num_chains
write(f, " ├── chain_$i.csv\n")
end
write(f, " Total: $(get(metadata, "num_chains", num_chains)) chains × " *
"$(get(metadata, "num_samples", "?")) samples\n")
write(f, " Size: $(round(total_size_gb, digits=2)) GB\n\n")
write(f, "data/\n")
write(f, " ├── text_data.csv\n")
write(f, " ├── expert_dim.csv\n")
write(f, " ├── expert_lr.csv\n")
write(f, " ├── segment_year_map.csv (V10)\n")
write(f, " ├── segment_info.csv (V10)\n")
write(f, " └── stan_data.json\n\n")
write(f, "=" ^ 78 * "\n")
write(f, "MODEL CONFIGURATION\n")
write(f, "=" ^ 78 * "\n\n")
write(f, "Chains: $(get(metadata, "num_chains", "?"))\n")
write(f, "Warmup: $(get(metadata, "num_warmup", "?"))\n")
write(f, "Samples: $(get(metadata, "num_samples", "?"))\n")
write(f, "Adapt delta: $(get(metadata, "adapt_delta", "?"))\n")
write(f, "Max depth: $(get(metadata, "max_depth", "?"))\n\n")
write(f, "Dimensions: $(join(get(metadata, "dimensions", ["?"]), ", "))\n\n")
write(f, "=" ^ 78 * "\n")
write(f, "HOW TO USE THESE RESULTS\n")
write(f, "=" ^ 78 * "\n\n")
write(f, "To extract party positions:\n\n")
write(f, " julia 02_post_estimation.jl\n\n")
write(f, "This will read the CSV files and generate party_positions_[timestamp].csv\n")
write(f, "with uncertainty estimates (SE, credible intervals).\n\n")
write(f, "=" ^ 78 * "\n")
write(f, "Generated: $(Dates.format(Dates.now(), "yyyy-mm-dd HH:MM:SS"))\n")
write(f, "=" ^ 78 * "\n")
end
end
"""
Wrapper for robust_save_model_csv that handles the stanmodel tuple from run_4dim_stan_model.
The model execution now saves chains BEFORE returning, so this function:
1. Verifies chains are already saved in final_run_dir
2. Adds metadata and data files
3. Returns the output path
Arguments:
- stanmodel_tuple: Named tuple from run_4dim_stan_model (contains .stanmodel, .final_run_dir, etc.)
- model_data: Dict with data_dict, original data, and metadata
- output_dir: Base output directory (ignored - uses stanmodel_tuple.final_run_dir)
- compress: Ignored (CSV-first architecture)
- keep_local_backups: Ignored (chains already saved)
"""
function robust_save_model(
stanmodel_tuple,
model_data::Dict,
output_dir::String;
compress::Bool=true,
keep_local_backups::Int=2
)
# Extract the final run directory from the stanmodel tuple
if !hasproperty(stanmodel_tuple, :final_run_dir)
error("stanmodel_tuple missing :final_run_dir - chains may not be saved!")
end
final_run_dir = stanmodel_tuple.final_run_dir
# Prepare original data for saving
original_data = Dict{String, Any}()
if haskey(model_data, "manifesto")
original_data["text_data"] = model_data["manifesto"]
end
if haskey(model_data, "expert_dim")
original_data["expert_dim"] = model_data["expert_dim"]
end
if haskey(model_data, "expert_lr")
original_data["expert_lr"] = model_data["expert_lr"]
end
# V10: Save segment_year_map and segment_info for post-estimation
if haskey(model_data, "segment_year")
original_data["segment_year_map"] = model_data["segment_year"]
end
if haskey(model_data, "segment_info")
original_data["segment_info"] = model_data["segment_info"]
end
# V9 fallback: Save party_year_map for post-estimation (includes interpolated years)
if haskey(model_data, "party_year")
original_data["party_year_map"] = model_data["party_year"]
end
# Prepare metadata
metadata = Dict{String, Any}()
if haskey(model_data, "model_info")
for (k, v) in model_data["model_info"]
metadata[string(k)] = v
end
end
# Get data_dict
data_dict = get(model_data, "data_dict", Dict{String, Any}())
# Call the CSV-first save with chains_already_saved=true
result = robust_save_model_csv(
final_run_dir,
data_dict,
original_data,
metadata;
chains_already_saved=true
)
return result.run_dir
end
end # module
@@ -0,0 +1,150 @@
#!/usr/bin/env julia
#############################################################################
## performance_monitoring.jl
## Post-run performance diagnostics for Stan sampling jobs
#############################################################################
using StanSample
using Statistics: mean, median
using Dates
function _safe_column(df, candidates::Vector{String})
for candidate in candidates
sym = Symbol(candidate)
if sym in names(df)
return df[!, sym]
elseif candidate in names(df)
return df[!, candidate]
end
end
return nothing
end
function _clean_values(vec)
cleaned = Float64[]
for v in vec
if v isa Missing || v === nothing
continue
end
try
value = Float64(v)
isfinite(value) && push!(cleaned, value)
catch
continue
end
end
return cleaned
end
function _summarize_vector(vec)
if vec === nothing
return Dict{String,Any}("available" => false)
end
cleaned = _clean_values(vec)
if isempty(cleaned)
return Dict{String,Any}("available" => false)
end
return Dict{String,Any}(
"available" => true,
"count" => length(cleaned),
"mean" => mean(cleaned),
"median" => median(cleaned),
"min" => minimum(cleaned),
"max" => maximum(cleaned)
)
end
function monitor_sampling_performance!(stanmodel;
run_metrics::Union{Nothing,Dict{String,Any}}=nothing,
metrics_path::Union{Nothing,String}=nothing,
csv_paths::Union{Nothing,Vector{String}}=nothing,
aggregate_metrics::Union{Nothing,Dict{String,Any}}=nothing,
max_depth::Union{Nothing,Int}=nothing)
csv_paths === nothing && (csv_paths = discover_stan_csvs([stanmodel.tmpdir]))
aggregate_metrics === nothing && begin
_, aggregate_metrics = collect_run_metrics(csv_paths; max_depth=max_depth)
end
summary_df = nothing
try
summary_df = read_summary(stanmodel)
catch e
println("Warning: could not read Stan summary: $e")
end
performance = Dict{String,Any}(
"generated_at" => Dates.format(Dates.now(), "yyyy-mm-ddTHH:MM:SS"),
"csv_paths" => csv_paths
)
if summary_df !== nothing
performance["ess_bulk"] = _summarize_vector(_safe_column(summary_df, ["ess_bulk", "ess"]))
performance["ess_tail"] = _summarize_vector(_safe_column(summary_df, ["ess_tail"]))
performance["ess_per_sec"] = _summarize_vector(_safe_column(summary_df, ["ess_per_sec", "n_eff/s"]))
performance["r_hat"] = _summarize_vector(_safe_column(summary_df, ["r_hat"]))
if haskey(performance["r_hat"], "available") && performance["r_hat"]["available"]
performance["r_hat"]["max"] = maximum(_clean_values(_safe_column(summary_df, ["r_hat"])))
end
performance["parameters_considered"] = size(summary_df, 1)
else
performance["ess_bulk"] = Dict{String,Any}("available" => false)
performance["ess_tail"] = Dict{String,Any}("available" => false)
performance["ess_per_sec"] = Dict{String,Any}("available" => false)
performance["r_hat"] = Dict{String,Any}("available" => false)
performance["parameters_considered"] = 0
end
divergences = get(aggregate_metrics, "divergences", 0)
total_samples = stanmodel.num_samples * stanmodel.num_chains
divergence_rate = total_samples > 0 ? divergences / total_samples : nothing
performance["divergences"] = Dict{String,Any}(
"count" => divergences,
"rate" => divergence_rate,
"total_draws" => total_samples
)
performance["leapfrog"] = Dict{String,Any}(
"mean" => get(aggregate_metrics, "mean_leapfrog", nothing)
)
performance["step_size"] = Dict{String,Any}(
"mean" => get(aggregate_metrics, "mean_step_size", nothing)
)
sampling_seconds = get(aggregate_metrics, "sampling_seconds", nothing)
if sampling_seconds !== nothing && sampling_seconds > 0
performance["throughput"] = Dict{String,Any}(
"samples_per_second" => (total_samples / sampling_seconds),
"seconds_sampling" => sampling_seconds
)
else
performance["throughput"] = Dict{String,Any}(
"samples_per_second" => nothing,
"seconds_sampling" => sampling_seconds
)
end
if run_metrics !== nothing
run_metrics["performance"] = performance
if metrics_path !== nothing
safe_write_json(metrics_path, run_metrics)
end
elseif metrics_path !== nothing
temp_metrics = Dict{String,Any}("performance" => performance)
safe_write_json(metrics_path, temp_metrics)
end
println("\nPERFORMANCE SUMMARY")
println(" ESS bulk (mean): $(get(performance["ess_bulk"], "mean", "n/a"))")
println(" ESS/sec (mean): $(get(performance["ess_per_sec"], "mean", "n/a"))")
println(" Divergences: $(divergences)")
println(" Divergence rate: $(divergence_rate === nothing ? "n/a" : round(divergence_rate, digits=6))")
println(" Mean leapfrog steps: $(get(performance["leapfrog"], "mean", "n/a"))")
return performance
end
+450
View File
@@ -0,0 +1,450 @@
#!/usr/bin/env julia
#############################################################################
## validate_construct.jl
## Construct validity: Party family ordering and temporal stability
##
## Following Claassen (2019), this script validates:
## 1. Party family ordering: Do family means follow theoretically expected orderings?
## 2. Temporal stability: Flag parties with implausible position changes
##
## Uses ParlGov party family classifications (Döring & Manow 2024) via PartyFacts IDs.
#############################################################################
using CSV, DataFrames, Statistics, StatsBase, Dates, Printf
# Family code → display name mapping
const FAMILY_DISPLAY_NAMES = Dict(
"com" => "Communist/Far Left",
"eco" => "Green/Ecological",
"soc" => "Social Democratic",
"lib" => "Liberal",
"chr" => "Christian Democratic",
"con" => "Conservative",
"right" => "Radical Right"
)
# Substantive families (drop Specialist, Other, Agrarian — heterogeneous or ambiguous)
const SUBSTANTIVE_FAMILIES = Set(["com", "eco", "soc", "lib", "chr", "con", "right"])
# Expected orderings (theoretically motivated)
# Economic: Communist < Social Democratic < Green < Christian Democratic < Conservative
# (5-family core — Liberal position is ambiguous cross-nationally)
const EXPECTED_ECONOMIC_ORDER = ["com", "soc", "eco", "chr", "con"]
# Cultural: Green < Liberal < Social Democratic < Christian Democratic < Conservative < Radical Right
const EXPECTED_GALTAN_ORDER = ["eco", "lib", "soc", "chr", "con", "right"]
function load_model_output(base_dir::String=".")
"""Load the most recent 2D model party positions output"""
position_files = filter(f -> startswith(f, "party_positions_") && endswith(f, ".csv") &&
!endswith(f, "_metadata.txt") && !endswith(f, "_tables.tex"), readdir(base_dir))
legacy_files = filter(f -> startswith(f, "party_positions_v1_") && endswith(f, ".csv"), readdir(base_dir))
append!(position_files, legacy_files)
if !isempty(position_files)
latest = sort(position_files)[end]
println("Loading model output: $latest")
return CSV.read(joinpath(base_dir, latest), DataFrame), latest
end
# Check output estimations directory
est_dir = joinpath(base_dir, "outputs", "estimations", "latest")
if isdir(est_dir)
est_files = filter(f -> startswith(f, "party_positions_") && endswith(f, ".csv") &&
!endswith(f, "_metadata.txt") && !endswith(f, "_tables.tex"), readdir(est_dir))
if !isempty(est_files)
latest = sort(est_files)[end]
println("Loading model output: outputs/estimations/latest/$latest")
return CSV.read(joinpath(est_dir, latest), DataFrame), latest
end
end
error("No party_positions_*.csv found. Run 02_post_estimation.jl first.")
end
function validate_party_families(model::DataFrame)
"""Check whether party family means follow theoretically expected orderings"""
println("\n" * "="^60)
println("PARTY FAMILY ORDERING VALIDATION")
println("="^60)
println("\nUsing ParlGov family classifications (Döring & Manow 2024)")
println()
party_col = hasproperty(model, :party_id) ? :party_id : :party
# Load party families
families_df = CSV.read("data/party_families.csv", DataFrame)
# Join to model output
model_with_families = innerjoin(model, families_df, on=party_col => :partyfacts_id)
# Filter to substantive families
filter!(r -> r.family in SUBSTANTIVE_FAMILIES, model_with_families)
n_parties = length(unique(model_with_families[!, party_col]))
n_obs = nrow(model_with_families)
println(" Matched $n_parties parties ($n_obs party-years) across $(length(SUBSTANTIVE_FAMILIES)) families")
println()
# Compute family means
family_stats = combine(groupby(model_with_families, :family)) do df
DataFrame(
n_parties = length(unique(df[!, party_col])),
n_obs = nrow(df),
mean_economic = mean(df.economic_lr),
sd_economic = std(df.economic_lr),
mean_galtan = mean(df.galtan),
sd_galtan = std(df.galtan)
)
end
# Add display names
family_stats.family_name = [get(FAMILY_DISPLAY_NAMES, f, f) for f in family_stats.family]
# Sort by economic mean for display
sort!(family_stats, :mean_economic)
# Print table
@printf(" %-22s %7s %7s %10s %10s %10s %10s\n",
"Family", "Parties", "Obs", "Econ Mean", "Econ SD", "Cult Mean", "Cult SD")
println(" " * "-"^76)
for row in eachrow(family_stats)
@printf(" %-22s %7d %7d %10.3f %10.3f %10.3f %10.3f\n",
row.family_name, row.n_parties, row.n_obs,
row.mean_economic, row.sd_economic, row.mean_galtan, row.sd_galtan)
end
# Compute Spearman rank correlations for expected orderings
println()
# Economic ordering
econ_lookup = Dict(row.family => row.mean_economic for row in eachrow(family_stats))
econ_observed = [econ_lookup[f] for f in EXPECTED_ECONOMIC_ORDER if haskey(econ_lookup, f)]
econ_expected_ranks = collect(1:length(econ_observed))
econ_observed_ranks = ordinalrank(econ_observed)
rho_econ = corspearman(Float64.(econ_expected_ranks), Float64.(econ_observed_ranks))
println(@sprintf(" Economic ordering (5-family core): Spearman ρ = %.3f", rho_econ))
econ_families_used = [f for f in EXPECTED_ECONOMIC_ORDER if haskey(econ_lookup, f)]
econ_names = [get(FAMILY_DISPLAY_NAMES, f, f) for f in econ_families_used]
println(" Expected: ", join(econ_names, " < "))
observed_econ_order = econ_families_used[sortperm(econ_observed)]
observed_econ_names = [get(FAMILY_DISPLAY_NAMES, f, f) for f in observed_econ_order]
println(" Observed: ", join(observed_econ_names, " < "))
# Cultural ordering
galtan_lookup = Dict(row.family => row.mean_galtan for row in eachrow(family_stats))
galtan_observed = [galtan_lookup[f] for f in EXPECTED_GALTAN_ORDER if haskey(galtan_lookup, f)]
galtan_expected_ranks = collect(1:length(galtan_observed))
galtan_observed_ranks = ordinalrank(galtan_observed)
rho_galtan = corspearman(Float64.(galtan_expected_ranks), Float64.(galtan_observed_ranks))
println(@sprintf(" Cultural ordering (6-family): Spearman ρ = %.3f", rho_galtan))
galtan_families_used = [f for f in EXPECTED_GALTAN_ORDER if haskey(galtan_lookup, f)]
galtan_names = [get(FAMILY_DISPLAY_NAMES, f, f) for f in galtan_families_used]
println(" Expected: ", join(galtan_names, " < "))
observed_galtan_order = galtan_families_used[sortperm(galtan_observed)]
observed_galtan_names = [get(FAMILY_DISPLAY_NAMES, f, f) for f in observed_galtan_order]
println(" Observed: ", join(observed_galtan_names, " < "))
println()
println("-"^60)
if rho_econ >= 0.9 && rho_galtan >= 0.8
println("EXCELLENT: Family means follow expected orderings on both dimensions")
elseif rho_econ >= 0.7 && rho_galtan >= 0.7
println("GOOD: Family means broadly follow expected orderings")
else
println("CONCERN: Inspect family ordering results")
end
return family_stats, rho_econ, rho_galtan
end
function validate_temporal_stability(model::DataFrame)
"""Check for implausible year-to-year position changes"""
println("\n" * "="^60)
println("TEMPORAL STABILITY VALIDATION")
println("="^60)
println("\nFlagging parties with >0.10 change per year")
println()
party_col = hasproperty(model, :party_id) ? :party_id : :party
# Compute year-to-year changes within each party
sort!(model, [party_col, :year])
unstable_parties = []
for party_df in groupby(model, party_col)
if nrow(party_df) < 2
continue
end
party_id = party_df[1, party_col]
country = party_df[1, :country]
# Compute differences
for dim in [:economic_lr, :galtan]
vals = party_df[!, dim]
years = party_df.year
for i in 2:length(vals)
diff = abs(vals[i] - vals[i-1])
year_gap = years[i] - years[i-1]
# Normalize by year gap (handle multi-year gaps)
annual_change = diff / max(year_gap, 1)
if annual_change > 0.10
push!(unstable_parties, (
party_id = party_id,
country = country,
dimension = string(dim),
year_from = years[i-1],
year_to = years[i],
val_from = vals[i-1],
val_to = vals[i],
change = diff,
annual_change = annual_change
))
end
end
end
end
if isempty(unstable_parties)
println(" No parties with >0.10 annual change found")
println(" EXCELLENT: Positions are temporally stable")
return DataFrame()
end
unstable_df = DataFrame(unstable_parties)
sort!(unstable_df, :annual_change, rev=true)
println(" Found $(nrow(unstable_df)) instances of rapid change:")
println()
@printf(" %-8s %-8s %-12s %-10s %-10s %8s\n",
"Party", "Country", "Dimension", "Years", "Change", "Annual")
println(" " * "-"^60)
for row in eachrow(unstable_df[1:min(20, nrow(unstable_df)), :])
@printf(" %-8d %-8s %-12s %d->%d %8.3f %8.3f\n",
row.party_id, row.country, row.dimension,
row.year_from, row.year_to, row.change, row.annual_change)
end
if nrow(unstable_df) > 20
println(" ... and $(nrow(unstable_df) - 20) more")
end
println()
println("-"^60)
n_parties = length(unique(unstable_df.party_id))
n_total = length(unique(model[!, party_col]))
println(@sprintf("Unstable parties: %d/%d (%.1f%%)", n_parties, n_total, 100*n_parties/n_total))
return unstable_df
end
function validate_position_distributions(model::DataFrame)
"""Check overall distribution of positions makes sense"""
println("\n" * "="^60)
println("POSITION DISTRIBUTION VALIDATION")
println("="^60)
println("\nSummary statistics for model estimates")
println()
for dim in [:economic_lr, :galtan]
if !hasproperty(model, dim)
continue
end
vals = model[!, dim]
println("$dim:")
println(@sprintf(" Mean: %.3f (should be ~0.50)", mean(vals)))
println(@sprintf(" Median: %.3f (should be ~0.50)", median(vals)))
println(@sprintf(" SD: %.3f (should be ~0.15)", std(vals)))
println(@sprintf(" Min: %.3f", minimum(vals)))
println(@sprintf(" Max: %.3f", maximum(vals)))
println(@sprintf(" Q25: %.3f", quantile(vals, 0.25)))
println(@sprintf(" Q75: %.3f", quantile(vals, 0.75)))
println()
end
# Check for extreme values
println("Extreme positions (< 0.10 or > 0.90):")
party_col = hasproperty(model, :party_id) ? :party_id : :party
for dim in [:economic_lr, :galtan]
if !hasproperty(model, dim)
continue
end
extreme = filter(row -> row[dim] < 0.10 || row[dim] > 0.90, model)
n_extreme = nrow(extreme)
pct_extreme = 100 * n_extreme / nrow(model)
println(@sprintf(" %s: %d (%.1f%%)", dim, n_extreme, pct_extreme))
if n_extreme > 0 && n_extreme <= 10
for row in eachrow(extreme[1:min(5, nrow(extreme)), :])
println(@sprintf(" Party %d (%s) %d: %.3f",
row[party_col], row.country, row.year, row[dim]))
end
end
end
end
function validate_country_patterns(model::DataFrame)
"""Check country-level patterns make sense"""
println("\n" * "="^60)
println("COUNTRY-LEVEL VALIDATION")
println("="^60)
println("\nMean positions by country (should vary but not wildly)")
println()
country_stats = combine(groupby(model, :country)) do df
DataFrame(
n_parties = length(unique(hasproperty(df, :party_id) ? df.party_id : df.party)),
n_obs = nrow(df),
mean_econ = mean(df.economic_lr),
mean_galtan = mean(df.galtan),
sd_econ = std(df.economic_lr),
sd_galtan = std(df.galtan)
)
end
sort!(country_stats, :n_obs, rev=true)
@printf("%-4s %6s %6s %8s %8s %8s %8s\n",
"CC", "Parties", "N", "Econ", "SD", "Cult", "SD")
println("-"^60)
for row in eachrow(country_stats[1:min(20, nrow(country_stats)), :])
@printf("%-4s %6d %6d %8.3f %8.3f %8.3f %8.3f\n",
row.country, row.n_parties, row.n_obs,
row.mean_econ, row.sd_econ, row.mean_galtan, row.sd_galtan)
end
# Flag countries with unusual patterns
println("\nCountries with unusual patterns:")
unusual = filter(row -> row.mean_econ < 0.35 || row.mean_econ > 0.65 ||
row.mean_galtan < 0.35 || row.mean_galtan > 0.65, country_stats)
if nrow(unusual) == 0
println(" None - all countries have balanced party systems")
else
for row in eachrow(unusual)
issues = String[]
if row.mean_econ < 0.35
push!(issues, "left-skewed economy")
elseif row.mean_econ > 0.65
push!(issues, "right-skewed economy")
end
if row.mean_galtan < 0.35
push!(issues, "cosmopolitan-skewed")
elseif row.mean_galtan > 0.65
push!(issues, "traditionalist-skewed")
end
println(" $(row.country): $(join(issues, ", "))")
end
end
return country_stats
end
function save_construct_results(families::DataFrame, unstable::DataFrame,
countries::DataFrame, output_dir::String="outputs/checks")
"""Save construct validation results"""
if !isdir(output_dir)
mkpath(output_dir)
end
timestamp = Dates.format(now(), "yyyy-mm-dd_HH-MM-SS")
if nrow(families) > 0
families_file = joinpath(output_dir, "construct_families_$timestamp.csv")
CSV.write(families_file, families)
println("\nSaved: $families_file")
end
if nrow(unstable) > 0
unstable_file = joinpath(output_dir, "construct_unstable_$timestamp.csv")
CSV.write(unstable_file, unstable)
println("Saved: $unstable_file")
end
if nrow(countries) > 0
country_file = joinpath(output_dir, "construct_countries_$timestamp.csv")
CSV.write(country_file, countries)
println("Saved: $country_file")
end
end
# Main execution
function main()
println("="^60)
println("CONSTRUCT VALIDITY: Face Validity Checks")
println("="^60)
println("Checking if model estimates match expectations")
println()
# Load data
model, model_file = load_model_output()
println("\nModel output:")
println(" Rows: $(nrow(model))")
println(" Columns: $(names(model))")
# Run validations
families, rho_econ, rho_galtan = validate_party_families(model)
unstable = validate_temporal_stability(model)
validate_position_distributions(model)
countries = validate_country_patterns(model)
# Save results
save_construct_results(families, unstable, countries)
# Summary
println("\n" * "="^60)
println("CONSTRUCT VALIDITY SUMMARY")
println("="^60)
println(@sprintf(" Party family ordering:"))
println(@sprintf(" Economic (5-family): Spearman ρ = %.3f", rho_econ))
println(@sprintf(" Cultural (6-family): Spearman ρ = %.3f", rho_galtan))
n_unstable = nrow(unstable) > 0 ? length(unique(unstable.party_id)) : 0
party_col = hasproperty(model, :party_id) ? :party_id : :party
n_total = length(unique(model[!, party_col]))
println(@sprintf(" Temporal stability: %d/%d parties stable (>0.10/yr threshold)",
n_total - n_unstable, n_total))
if rho_econ >= 0.9 && rho_galtan >= 0.8 && n_unstable < 0.1 * n_total
println("\n EXCELLENT: Model has good construct validity")
elseif rho_econ >= 0.7 && rho_galtan >= 0.7
println("\n GOOD: Model has reasonable construct validity")
else
println("\n CONCERN: Inspect family ordering results")
end
println("\n" * "="^60)
println("VALIDATION COMPLETE")
println("="^60)
return (families=families, unstable=unstable, countries=countries)
end
if abspath(PROGRAM_FILE) == @__FILE__
main()
end
+533
View File
@@ -0,0 +1,533 @@
#!/usr/bin/env julia
#############################################################################
## validate_convergent.jl
## Convergent validity: Compare model estimates to external expert surveys
##
## Following Claassen (2019), this script computes:
## - Pearson/Spearman correlations between model and expert estimates
## - Fisher z-transformation for correlation confidence intervals
## - Mean Absolute Error (MAE) and Root Mean Square Error (RMSE)
## - Breakdown by survey project and decade
##
## Target: r > 0.8 with CHES (Claassen achieved 0.50-0.57)
#############################################################################
using CSV, DataFrames, Statistics, StatsBase, Dates, Printf, JSON
# Fisher z-transformation for correlation confidence intervals
fisher_z(r) = 0.5 * log((1 + r) / (1 - r))
fisher_z_inv(z) = (exp(2z) - 1) / (exp(2z) + 1)
function correlation_ci(r, n; alpha=0.05)
"""Calculate confidence interval for correlation using Fisher z-transformation"""
if n < 4
return (lower=NaN, upper=NaN)
end
z = fisher_z(r)
se = 1 / sqrt(n - 3)
z_crit = 1.96 # For 95% CI
z_lower = z - z_crit * se
z_upper = z + z_crit * se
return (lower=fisher_z_inv(z_lower), upper=fisher_z_inv(z_upper))
end
function load_model_output(base_dir::String=".")
"""Load the most recent 2D model party positions output"""
# First check for post_estimation output in root (current name, with legacy fallback)
position_files = filter(f -> startswith(f, "party_positions_") && endswith(f, ".csv") &&
!endswith(f, "_metadata.txt") && !endswith(f, "_tables.tex"), readdir(base_dir))
legacy_files = filter(f -> startswith(f, "party_positions_v1_") && endswith(f, ".csv"), readdir(base_dir))
append!(position_files, legacy_files)
if !isempty(position_files)
latest = sort(position_files)[end]
println("Loading model output: $latest")
return CSV.read(joinpath(base_dir, latest), DataFrame), latest
end
# Check current pipeline output directory, with legacy estimations/ fallback
for (label, est_dir) in [
("outputs/estimations/latest", joinpath(base_dir, "outputs", "estimations", "latest")),
("estimations", joinpath(base_dir, "estimations")),
]
if isdir(est_dir)
est_files = filter(f -> startswith(f, "party_positions_") && endswith(f, ".csv") &&
!endswith(f, "_metadata.txt") && !endswith(f, "_tables.tex"), readdir(est_dir))
if !isempty(est_files)
latest = sort(est_files)[end]
println("Loading model output: $label/$latest")
return CSV.read(joinpath(est_dir, latest), DataFrame), latest
end
end
end
error("No party_positions_*.csv found. Run 02_post_estimation.jl first.")
end
function load_expert_data(base_dir::String=".")
"""Load expert survey data files"""
data_dir = isfile(joinpath(base_dir, "expert.csv")) ? base_dir : joinpath(base_dir, "data")
# Load dimension-specific expert data
expert_file = joinpath(data_dir, "expert.csv")
if !isfile(expert_file)
error("expert.csv not found in $base_dir or $(joinpath(base_dir, "data"))")
end
println("Loading expert.csv...")
expert = CSV.read(expert_file, DataFrame)
println(" Rows: $(nrow(expert))")
println(" Variables: $(unique(expert.var))")
# Load L-R data
lr_file = joinpath(data_dir, "lr_data.csv")
if !isfile(lr_file)
error("lr_data.csv not found in $base_dir or $(joinpath(base_dir, "data"))")
end
println("Loading lr_data.csv...")
lr_data = CSV.read(lr_file, DataFrame)
println(" Rows: $(nrow(lr_data))")
println(" Variables: $(unique(lr_data.var))")
return expert, lr_data
end
function validate_economic_lr(model::DataFrame, expert::DataFrame)
"""Validate economic_lr against CHES/V-Party/POPPA/GPS lrecon"""
println("\n" * "="^60)
println("CONVERGENT VALIDITY: economic_lr")
println("="^60)
# Filter expert data for economic dimension
econ_vars = filter(v -> startswith(v, "lrecon_"), unique(expert.var))
econ_expert = filter(row -> row.var in econ_vars, expert)
println("\nExpert data variables: $(econ_vars)")
println("Expert observations: $(nrow(econ_expert))")
# Merge with model output
# Model has party_id column (from segment-based), expert has party column
if hasproperty(model, :party_id)
model_merge = select(model, :party_id => :party, :year, :economic_lr, :economic_lr_se)
else
model_merge = select(model, :party, :year, :economic_lr, :economic_lr_se)
end
merged = innerjoin(econ_expert, model_merge, on=[:party, :year])
println("Merged observations: $(nrow(merged))")
if nrow(merged) < 10
println("WARNING: Too few observations for meaningful validation")
return nothing
end
# Compute overall correlation
r_pearson = cor(merged.val, merged.economic_lr)
r_spearman = corspearman(merged.val, merged.economic_lr)
mae = mean(abs.(merged.val .- merged.economic_lr))
rmse = sqrt(mean((merged.val .- merged.economic_lr).^2))
ci = correlation_ci(r_pearson, nrow(merged))
println("\n--- Overall Statistics ---")
println(@sprintf(" Pearson r: %.4f [%.4f, %.4f]", r_pearson, ci.lower, ci.upper))
println(@sprintf(" Spearman r: %.4f", r_spearman))
println(@sprintf(" MAE: %.4f", mae))
println(@sprintf(" RMSE: %.4f", rmse))
println(@sprintf(" N: %d", nrow(merged)))
# Breakdown by project
println("\n--- By Project ---")
by_project = combine(groupby(merged, :project)) do df
n = nrow(df)
if n < 3
return DataFrame(n=n, r_pearson=NaN, mae=NaN)
end
DataFrame(
n = n,
r_pearson = cor(df.val, df.economic_lr),
r_spearman = corspearman(df.val, df.economic_lr),
mae = mean(abs.(df.val .- df.economic_lr)),
rmse = sqrt(mean((df.val .- df.economic_lr).^2))
)
end
for row in eachrow(sort(by_project, :n, rev=true))
if !isnan(row.r_pearson)
println(@sprintf(" %-10s: r=%.3f, MAE=%.3f, n=%d",
row.project, row.r_pearson, row.mae, row.n))
end
end
# Breakdown by decade
println("\n--- By Decade ---")
merged.decade = div.(merged.year, 10) .* 10
by_decade = combine(groupby(merged, :decade)) do df
n = nrow(df)
if n < 3
return DataFrame(n=n, r_pearson=NaN, mae=NaN)
end
DataFrame(
n = n,
r_pearson = cor(df.val, df.economic_lr),
mae = mean(abs.(df.val .- df.economic_lr))
)
end
for row in eachrow(sort(by_decade, :decade))
if !isnan(row.r_pearson)
println(@sprintf(" %ds: r=%.3f, MAE=%.3f, n=%d",
row.decade, row.r_pearson, row.mae, row.n))
end
end
return (
dimension = "economic_lr",
r_pearson = r_pearson,
r_spearman = r_spearman,
ci_lower = ci.lower,
ci_upper = ci.upper,
mae = mae,
rmse = rmse,
n = nrow(merged),
by_project = by_project,
by_decade = by_decade
)
end
function validate_galtan(model::DataFrame, expert::DataFrame)
"""Validate cultural cosmopolitan--traditionalist estimates against CHES and V-Party/GPS cultural measures"""
println("\n" * "="^60)
println("CONVERGENT VALIDITY: cultural cosmopolitan--traditionalist")
println("="^60)
# Filter expert data for cultural cosmopolitan--traditionalist dimension
galtan_vars = filter(v -> occursin("galtan", v) || occursin("libcon", v), unique(expert.var))
galtan_expert = filter(row -> row.var in galtan_vars, expert)
println("\nExpert data variables: $(galtan_vars)")
println("Expert observations: $(nrow(galtan_expert))")
# Merge with model output
if hasproperty(model, :party_id)
model_merge = select(model, :party_id => :party, :year, :galtan, :galtan_se)
else
model_merge = select(model, :party, :year, :galtan, :galtan_se)
end
merged = innerjoin(galtan_expert, model_merge, on=[:party, :year])
println("Merged observations: $(nrow(merged))")
if nrow(merged) < 10
println("WARNING: Too few observations for meaningful validation")
return nothing
end
# Compute overall correlation
r_pearson = cor(merged.val, merged.galtan)
r_spearman = corspearman(merged.val, merged.galtan)
mae = mean(abs.(merged.val .- merged.galtan))
rmse = sqrt(mean((merged.val .- merged.galtan).^2))
ci = correlation_ci(r_pearson, nrow(merged))
println("\n--- Overall Statistics ---")
println(@sprintf(" Pearson r: %.4f [%.4f, %.4f]", r_pearson, ci.lower, ci.upper))
println(@sprintf(" Spearman r: %.4f", r_spearman))
println(@sprintf(" MAE: %.4f", mae))
println(@sprintf(" RMSE: %.4f", rmse))
println(@sprintf(" N: %d", nrow(merged)))
# Breakdown by project
println("\n--- By Project ---")
by_project = combine(groupby(merged, :project)) do df
n = nrow(df)
if n < 3
return DataFrame(n=n, r_pearson=NaN, mae=NaN)
end
DataFrame(
n = n,
r_pearson = cor(df.val, df.galtan),
r_spearman = corspearman(df.val, df.galtan),
mae = mean(abs.(df.val .- df.galtan)),
rmse = sqrt(mean((df.val .- df.galtan).^2))
)
end
for row in eachrow(sort(by_project, :n, rev=true))
if !isnan(row.r_pearson)
println(@sprintf(" %-10s: r=%.3f, MAE=%.3f, n=%d",
row.project, row.r_pearson, row.mae, row.n))
end
end
# Breakdown by decade
println("\n--- By Decade ---")
merged.decade = div.(merged.year, 10) .* 10
by_decade = combine(groupby(merged, :decade)) do df
n = nrow(df)
if n < 3
return DataFrame(n=n, r_pearson=NaN, mae=NaN)
end
DataFrame(
n = n,
r_pearson = cor(df.val, df.galtan),
mae = mean(abs.(df.val .- df.galtan))
)
end
for row in eachrow(sort(by_decade, :decade))
if !isnan(row.r_pearson)
println(@sprintf(" %ds: r=%.3f, MAE=%.3f, n=%d",
row.decade, row.r_pearson, row.mae, row.n))
end
end
return (
dimension = "galtan",
r_pearson = r_pearson,
r_spearman = r_spearman,
ci_lower = ci.lower,
ci_upper = ci.upper,
mae = mae,
rmse = rmse,
n = nrow(merged),
by_project = by_project,
by_decade = by_decade
)
end
function validate_discriminant(model::DataFrame, expert::DataFrame)
"""Compute cross-dimension correlations for discriminant validity (Campbell & Fiske 1959 MTMM)"""
println("\n" * "="^60)
println("DISCRIMINANT VALIDITY: Cross-dimension correlations")
println("="^60)
println("\nCampbell & Fiske (1959) MTMM framework:")
println(" Convergent: same dimension, different method → HIGH")
println(" Discriminant: different dimension, different method → LOW")
println()
# Get party column
if hasproperty(model, :party_id)
model_econ = select(model, :party_id => :party, :year, :economic_lr)
model_gal = select(model, :party_id => :party, :year, :galtan)
else
model_econ = select(model, :party, :year, :economic_lr)
model_gal = select(model, :party, :year, :galtan)
end
results = []
# 1. Expert economic vs Model economic (convergent - already computed, include for matrix)
econ_vars = filter(v -> startswith(v, "lrecon_"), unique(expert.var))
econ_expert = filter(row -> row.var in econ_vars, expert)
merged_ee = innerjoin(econ_expert, model_econ, on=[:party, :year])
if nrow(merged_ee) >= 10
r = cor(merged_ee.val, merged_ee.economic_lr)
push!(results, (model_dim="economic_lr", expert_dim="economic",
r_pearson=r, r_spearman=corspearman(merged_ee.val, merged_ee.economic_lr),
n=nrow(merged_ee), type="convergent"))
@printf(" Model Economic × Expert Economic: r = %.3f (convergent, n=%d)\n", r, nrow(merged_ee))
end
# 2. Expert economic vs model cultural dimension (discriminant)
merged_eg = innerjoin(econ_expert, model_gal, on=[:party, :year])
if nrow(merged_eg) >= 10
r = cor(merged_eg.val, merged_eg.galtan)
push!(results, (model_dim="galtan", expert_dim="economic",
r_pearson=r, r_spearman=corspearman(merged_eg.val, merged_eg.galtan),
n=nrow(merged_eg), type="discriminant"))
@printf(" Model Cultural × Expert Economic: r = %.3f (discriminant, n=%d)\n", r, nrow(merged_eg))
end
# 3. Expert cultural vs model cultural dimension (convergent - already computed, include for matrix)
galtan_vars = filter(v -> occursin("galtan", v) || occursin("libcon", v), unique(expert.var))
galtan_expert = filter(row -> row.var in galtan_vars, expert)
merged_gg = innerjoin(galtan_expert, model_gal, on=[:party, :year])
if nrow(merged_gg) >= 10
r = cor(merged_gg.val, merged_gg.galtan)
push!(results, (model_dim="galtan", expert_dim="galtan",
r_pearson=r, r_spearman=corspearman(merged_gg.val, merged_gg.galtan),
n=nrow(merged_gg), type="convergent"))
@printf(" Model Cultural × Expert Cultural: r = %.3f (convergent, n=%d)\n", r, nrow(merged_gg))
end
# 4. Expert cultural vs model economic dimension (discriminant)
merged_ge = innerjoin(galtan_expert, model_econ, on=[:party, :year])
if nrow(merged_ge) >= 10
r = cor(merged_ge.val, merged_ge.economic_lr)
push!(results, (model_dim="economic_lr", expert_dim="galtan",
r_pearson=r, r_spearman=corspearman(merged_ge.val, merged_ge.economic_lr),
n=nrow(merged_ge), type="discriminant"))
@printf(" Model Economic × Expert Cultural: r = %.3f (discriminant, n=%d)\n", r, nrow(merged_ge))
end
println()
println("MTMM Matrix:")
println(" Expert Economic Expert Cultural")
for r in results
if r.model_dim == "economic_lr" && r.expert_dim == "economic"
@printf(" Model Economic: %.3f ", r.r_pearson)
end
end
for r in results
if r.model_dim == "economic_lr" && r.expert_dim == "galtan"
@printf("%.3f\n", r.r_pearson)
end
end
for r in results
if r.model_dim == "galtan" && r.expert_dim == "economic"
@printf(" Model Cultural: %.3f ", r.r_pearson)
end
end
for r in results
if r.model_dim == "galtan" && r.expert_dim == "galtan"
@printf("%.3f\n", r.r_pearson)
end
end
return DataFrame(results)
end
function save_validation_results(results::Vector, output_dir::String="validation";
discriminant::Union{DataFrame, Nothing}=nothing)
"""Save validation results to CSV files"""
if !isdir(output_dir)
mkpath(output_dir)
end
timestamp = Dates.format(now(), "yyyy-mm-dd_HH-MM-SS")
# Summary table
summary_rows = []
for r in results
if r !== nothing
push!(summary_rows, (
dimension = r.dimension,
r_pearson = r.r_pearson,
r_spearman = r.r_spearman,
ci_lower = r.ci_lower,
ci_upper = r.ci_upper,
mae = r.mae,
rmse = r.rmse,
n = r.n
))
end
end
if !isempty(summary_rows)
summary_df = DataFrame(summary_rows)
summary_file = joinpath(output_dir, "convergent_summary_$timestamp.csv")
CSV.write(summary_file, summary_df)
println("\nSaved: $summary_file")
end
# By-project tables
for r in results
if r !== nothing && hasproperty(r, :by_project) && r.by_project !== nothing
project_file = joinpath(output_dir, "convergent_$(r.dimension)_by_project_$timestamp.csv")
CSV.write(project_file, r.by_project)
println("Saved: $project_file")
end
end
# By-decade tables
for r in results
if r !== nothing && hasproperty(r, :by_decade) && r.by_decade !== nothing
decade_file = joinpath(output_dir, "convergent_$(r.dimension)_by_decade_$timestamp.csv")
CSV.write(decade_file, r.by_decade)
println("Saved: $decade_file")
end
end
# Discriminant validity table
if discriminant !== nothing && nrow(discriminant) > 0
disc_file = joinpath(output_dir, "discriminant_summary_$timestamp.csv")
CSV.write(disc_file, discriminant)
println("Saved: $disc_file")
end
return summary_rows
end
function print_claassen_comparison(results::Vector)
"""Print comparison with Claassen (2019) benchmarks"""
println("\n" * "="^60)
println("COMPARISON WITH CLAASSEN (2019) BENCHMARKS")
println("="^60)
println("\nClaassen's results (mood estimates vs survey data):")
println(" Pearson r: 0.50-0.57")
println(" MAE: ~0.06 (6 pp on 0-1 scale)")
println()
println("Our target (party positions, should be HIGHER than mood):")
println(" Pearson r > 0.80 with expert surveys")
println(" MAE < 0.15 (reasonable measurement error)")
println()
println("-"^60)
@printf("%-15s %8s %8s %8s %8s\n", "Dimension", "r", "Target", "MAE", "Status")
println("-"^60)
for r in results
if r !== nothing
status = r.r_pearson > 0.80 ? "PASS" : (r.r_pearson > 0.70 ? "OK" : "LOW")
@printf("%-15s %8.3f %8s %8.3f %8s\n",
r.dimension, r.r_pearson, "> 0.80", r.mae, status)
end
end
println("-"^60)
end
# Main execution
function main()
println("="^60)
println("CONVERGENT VALIDITY: Model vs Expert Surveys")
println("="^60)
println("Following Claassen (2019) validation framework")
println()
# Load data
model, model_file = load_model_output()
expert, _ = load_expert_data()
println("\nModel output:")
println(" Rows: $(nrow(model))")
println(" Columns: $(names(model))")
# Run validations
results = []
push!(results, validate_economic_lr(model, expert))
push!(results, validate_galtan(model, expert))
# Run discriminant validity
discriminant = validate_discriminant(model, expert)
# Save results
save_validation_results(results; discriminant=discriminant)
# Print Claassen comparison
print_claassen_comparison(results)
println("\n" * "="^60)
println("VALIDATION COMPLETE")
println("="^60)
return results
end
if abspath(PROGRAM_FILE) == @__FILE__
main()
end
+375
View File
@@ -0,0 +1,375 @@
#!/usr/bin/env julia
#############################################################################
## validate_external.jl
## Out-of-sample validation via held-out expert observations
##
## Design:
## - Text data stays 100% intact (same parties, segments, indices)
## - 20% of expert/LR observations held out (stratified by source)
## - Expert-only parties (no text data) are never held out
## - Model trains on 80% expert + 100% text
## - Held-out expert ratings compared to model predictions
##
## Usage:
## julia scripts/validate_external.jl prepare
## julia 01_run_model.jl --data-dir validation/external_split/
## julia 02_post_estimation.jl # on training run
## julia scripts/validate_external.jl compute <model_positions.csv>
#############################################################################
using CSV, DataFrames, Statistics, Random, Dates, Printf
const HOLDOUT_FRAC = 0.20
const SEED = 42
# =========================================================================
# STEP 1: Prepare train/test split
# =========================================================================
function prepare_holdout_data(base_dir::String=".")
println("="^70)
println("PREPARING OUT-OF-SAMPLE VALIDATION SPLIT")
println("="^70)
println()
# Load data
text_data = CSV.read(joinpath(base_dir, "text_data.csv"), DataFrame)
expert = CSV.read(joinpath(base_dir, "expert.csv"), DataFrame)
lr_data = CSV.read(joinpath(base_dir, "lr_data.csv"), DataFrame)
println("Full dataset:")
println(" text_data: $(nrow(text_data)) rows, $(length(unique(text_data.party))) parties")
println(" expert: $(nrow(expert)) rows, $(length(unique(expert.party))) parties")
println(" lr_data: $(nrow(lr_data)) rows, $(length(unique(lr_data.party))) parties")
# Identify expert-only parties (no text data) — these are NEVER held out
text_parties = Set(unique(text_data.party))
expert_only_parties = Set(p for p in unique(vcat(expert.party, lr_data.party))
if !(p in text_parties))
println()
println("Expert-only parties (protected from holdout): $(length(expert_only_parties))")
# Split expert data: stratified by source variable
# Safeguard: ensure each party keeps at least one observation in training
Random.seed!(SEED)
expert.row_id = 1:nrow(expert)
expert.is_holdout = falses(nrow(expert))
for var_group in groupby(expert, :var)
var_name = first(var_group.var)
eligible = findall(row -> !(row.party in expert_only_parties), eachrow(var_group))
n_holdout = round(Int, length(eligible) * HOLDOUT_FRAC)
holdout_candidates = shuffle(eligible)
# Track per-party counts to ensure at least 1 stays in training
party_train_count = Dict{Int, Int}()
for idx in eligible
p = var_group.party[idx]
party_train_count[p] = get(party_train_count, p, 0) + 1
end
n_held = 0
for idx in holdout_candidates
n_held >= n_holdout && break
p = var_group.party[idx]
if party_train_count[p] > 1 # keep at least 1 in training
row_id = var_group.row_id[idx]
expert.is_holdout[row_id] = true
party_train_count[p] -= 1
n_held += 1
end
end
end
# Split LR data: stratified by source variable (same safeguard)
lr_data.row_id = 1:nrow(lr_data)
lr_data.is_holdout = falses(nrow(lr_data))
for var_group in groupby(lr_data, :var)
var_name = first(var_group.var)
eligible = findall(row -> !(row.party in expert_only_parties), eachrow(var_group))
n_holdout = round(Int, length(eligible) * HOLDOUT_FRAC)
holdout_candidates = shuffle(eligible)
party_train_count = Dict{Int, Int}()
for idx in eligible
p = var_group.party[idx]
party_train_count[p] = get(party_train_count, p, 0) + 1
end
n_held = 0
for idx in holdout_candidates
n_held >= n_holdout && break
p = var_group.party[idx]
if party_train_count[p] > 1
row_id = var_group.row_id[idx]
lr_data.is_holdout[row_id] = true
party_train_count[p] -= 1
n_held += 1
end
end
end
# Create train/test splits
expert_train = expert[.!expert.is_holdout, Not([:row_id, :is_holdout])]
expert_test = expert[expert.is_holdout, Not([:row_id, :is_holdout])]
lr_train = lr_data[.!lr_data.is_holdout, Not([:row_id, :is_holdout])]
lr_test = lr_data[lr_data.is_holdout, Not([:row_id, :is_holdout])]
# Report split
println()
println("Split summary ($(round(100*HOLDOUT_FRAC))% holdout):")
println(" expert train: $(nrow(expert_train)) rows ($(round(100*nrow(expert_train)/nrow(expert), digits=1))%)")
println(" expert test: $(nrow(expert_test)) rows ($(round(100*nrow(expert_test)/nrow(expert), digits=1))%)")
println(" lr train: $(nrow(lr_train)) rows ($(round(100*nrow(lr_train)/nrow(lr_data), digits=1))%)")
println(" lr test: $(nrow(lr_test)) rows ($(round(100*nrow(lr_test)/nrow(lr_data), digits=1))%)")
# Report per-source breakdown
println()
println("Per-source breakdown (expert):")
for var_name in sort(unique(expert.var))
n_full = count(expert.var .== var_name)
n_test = count(expert_test.var .== var_name)
println(" $var_name: $(n_full - n_test) train / $n_test test")
end
println()
println("Per-source breakdown (LR):")
for var_name in sort(unique(lr_data.var))
n_full = count(lr_data.var .== var_name)
n_test = count(lr_test.var .== var_name)
println(" $var_name: $(n_full - n_test) train / $n_test test")
end
# Verify: training set has same parties as full set
train_parties_expert = Set(unique(expert_train.party))
train_parties_lr = Set(unique(lr_train.party))
full_parties_expert = Set(unique(expert.party))
full_parties_lr = Set(unique(lr_data.party))
lost_expert = setdiff(full_parties_expert, train_parties_expert)
lost_lr = setdiff(full_parties_lr, train_parties_lr)
println()
if isempty(lost_expert) && isempty(lost_lr)
println("✓ No parties lost from training set")
else
println("⚠ Parties lost from expert training: $(length(lost_expert))")
println("⚠ Parties lost from LR training: $(length(lost_lr))")
end
# Save to output directory
# Files use standard names so 01_run_model.jl can load with --data-dir
output_dir = joinpath(base_dir, "validation", "external_split")
mkpath(output_dir)
# Training files (standard names for model loading)
CSV.write(joinpath(output_dir, "text_data.csv"), text_data) # UNCHANGED
CSV.write(joinpath(output_dir, "expert.csv"), expert_train)
CSV.write(joinpath(output_dir, "lr_data.csv"), lr_train)
# Test files (for compute step)
CSV.write(joinpath(output_dir, "expert_test.csv"), expert_test)
CSV.write(joinpath(output_dir, "lr_data_test.csv"), lr_test)
# Copy union mapping (needed by model)
if isdir(joinpath(base_dir, "data"))
mkpath(joinpath(output_dir, "data"))
cp(joinpath(base_dir, "data", "union_mapping.csv"),
joinpath(output_dir, "data", "union_mapping.csv"), force=true)
end
println()
println("Files saved to: $output_dir")
println(" text_data.csv — IDENTICAL to original ($(nrow(text_data)) rows)")
println(" expert.csv — training only ($(nrow(expert_train)) rows)")
println(" lr_data.csv — training only ($(nrow(lr_train)) rows)")
println(" expert_test.csv — held-out ($(nrow(expert_test)) rows)")
println(" lr_data_test.csv — held-out ($(nrow(lr_test)) rows)")
return output_dir
end
# =========================================================================
# STEP 2: Compute held-out validation metrics
# =========================================================================
function compute_holdout_metrics(model_file::String, test_dir::String)
println()
println("="^70)
println("COMPUTING HELD-OUT VALIDATION METRICS")
println("="^70)
# Load model output
model = CSV.read(model_file, DataFrame)
party_col = hasproperty(model, :party_id) ? :party_id : :party
println("Model output: $(nrow(model)) party-years")
# Load test data
expert_test = CSV.read(joinpath(test_dir, "expert_test.csv"), DataFrame)
lr_test = CSV.read(joinpath(test_dir, "lr_data_test.csv"), DataFrame)
println("Held-out expert: $(nrow(expert_test)) observations")
println("Held-out LR: $(nrow(lr_test)) observations")
# Build lookup: (party, year) → model estimates
model_lookup = Dict{Tuple{Int,Int}, NamedTuple}()
for row in eachrow(model)
key = (row[party_col], row.year)
model_lookup[key] = (
economic_lr = row.economic_lr,
galtan = row.galtan,
economic_lr_se = hasproperty(row, :economic_lr_se) ? row.economic_lr_se : missing,
galtan_se = hasproperty(row, :galtan_se) ? row.galtan_se : missing,
economic_lr_q025 = hasproperty(row, :economic_lr_q025) ? row.economic_lr_q025 : missing,
economic_lr_q975 = hasproperty(row, :economic_lr_q975) ? row.economic_lr_q975 : missing,
galtan_q025 = hasproperty(row, :galtan_q025) ? row.galtan_q025 : missing,
galtan_q975 = hasproperty(row, :galtan_q975) ? row.galtan_q975 : missing,
)
end
# Map expert variables to dimensions
econ_vars = Set(["lrecon_ches", "lrecon_poppa", "lrecon_gps", "lrecon_vparty", "welf_vparty"])
galtan_vars = Set(["galtan_ches", "libcon_gps", "immig_vparty", "lgbt_vparty",
"culsup_vparty", "relig_vparty", "gender_vparty"])
# Process expert test observations
results = NamedTuple[]
for row in eachrow(expert_test)
key = (row.party, row.year)
haskey(model_lookup, key) || continue
m = model_lookup[key]
if row.var in econ_vars
dim = "economic_lr"
model_val = m.economic_lr
model_q025 = m.economic_lr_q025
model_q975 = m.economic_lr_q975
elseif row.var in galtan_vars
dim = "galtan"
model_val = m.galtan
model_q025 = m.galtan_q025
model_q975 = m.galtan_q975
else
continue
end
covered = !ismissing(model_q025) && !ismissing(model_q975) &&
row.val >= model_q025 && row.val <= model_q975
push!(results, (
party = row.party,
country = row.country,
year = row.year,
var = row.var,
dimension = dim,
expert_val = row.val,
model_val = model_val,
error = row.val - model_val,
abs_error = abs(row.val - model_val),
covered_95 = ismissing(model_q025) ? missing : covered,
))
end
if isempty(results)
println("ERROR: No matching test observations found")
return nothing
end
results_df = DataFrame(results)
# Compute and report metrics
println()
println("-"^70)
println("HELD-OUT VALIDATION RESULTS")
println("-"^70)
# Overall
overall_r = cor(results_df.expert_val, results_df.model_val)
overall_mae = mean(results_df.abs_error)
overall_rmse = sqrt(mean(results_df.error .^ 2))
println()
println(@sprintf("Overall: r=%.4f, MAE=%.4f, RMSE=%.4f, n=%d",
overall_r, overall_mae, overall_rmse, nrow(results_df)))
# By dimension
println()
println("By dimension:")
for dim in sort(unique(results_df.dimension))
d = filter(r -> r.dimension == dim, results_df)
r_val = cor(d.expert_val, d.model_val)
mae = mean(d.abs_error)
rmse = sqrt(mean(d.error .^ 2))
cov = count(skipmissing(d.covered_95)) / count(!ismissing, d.covered_95)
println(@sprintf(" %-12s: r=%.4f, MAE=%.4f, RMSE=%.4f, CIC95=%.1f%%, n=%d",
dim, r_val, mae, rmse, 100*cov, nrow(d)))
end
# By source
println()
println("By source:")
for var in sort(unique(results_df.var))
v = filter(r -> r.var == var, results_df)
nrow(v) < 5 && continue
r_val = cor(v.expert_val, v.model_val)
mae = mean(v.abs_error)
println(@sprintf(" %-20s: r=%.4f, MAE=%.4f, n=%d", var, r_val, mae, nrow(v)))
end
return results_df
end
# =========================================================================
# Main
# =========================================================================
function main()
args = ARGS
if isempty(args) || args[1] == "prepare"
output_dir = prepare_holdout_data()
println()
println("="^70)
println("NEXT STEPS")
println("="^70)
println("""
1. Run model on training data:
julia 01_run_model.jl --data-dir $output_dir
2. Run post-estimation on training run output:
julia 02_post_estimation.jl
3. Compute held-out metrics:
julia scripts/validate_external.jl compute <party_positions_file.csv>
""")
elseif args[1] == "compute" && length(args) >= 2
model_file = args[2]
test_dir = joinpath("validation", "external_split")
isfile(model_file) || error("Model file not found: $model_file")
isdir(test_dir) || error("Test data not found. Run 'prepare' first.")
results_df = compute_holdout_metrics(model_file, test_dir)
if results_df !== nothing
timestamp = Dates.format(now(), "yyyy-mm-dd_HH-MM-SS")
output_file = joinpath("validation", "external_validation_$(timestamp).csv")
CSV.write(output_file, results_df)
println()
println("Detailed results saved: $output_file")
end
else
println("Usage:")
println(" julia scripts/validate_external.jl prepare")
println(" julia scripts/validate_external.jl compute <party_positions.csv>")
end
end
if abspath(PROGRAM_FILE) == @__FILE__
main()
end
+681
View File
@@ -0,0 +1,681 @@
#!/usr/bin/env julia
#############################################################################
## validate_uncertainty.jl
## Uncertainty validation: Posterior Predictive Coverage (PPC)
##
## Following Claassen (2019), this script computes:
## - PPC: What % of expert values fall within 95% posterior predictive interval?
## - Also computes 80% PPC for direct Claassen comparison (he reports 60.3%)
## - Wilson score CI for coverage proportion
## - Breakdown by survey source and decade
##
## Unlike credible interval coverage (which checks θ-CIs), posterior predictive
## coverage simulates what a new expert observation would look like given the
## model's beta likelihood, accounting for both position uncertainty AND
## measurement noise. Well-calibrated models should yield ~95% PPC at 95%.
##
## No model re-run needed: reads θ, γ, and φ from existing chain CSV files.
#############################################################################
using CSV, DataFrames, Statistics, Dates, Printf, Random, JSON
# =============================================================================
# Utility: Wilson score CI for a proportion
# =============================================================================
function wilson_ci(p, n; alpha=0.05)
if n == 0
return (lower=NaN, upper=NaN, se=NaN)
end
z = 1.96 # For 95% CI
denominator = 1 + z^2/n
center = (p + z^2/(2n)) / denominator
margin = z * sqrt((p*(1-p) + z^2/(4n))/n) / denominator
se = sqrt(p * (1-p) / n)
return (lower=center - margin, upper=center + margin, se=se)
end
# =============================================================================
# STEP 0: Find latest model run
# =============================================================================
function find_latest_run(base_dir::String="model_outputs")
if !isdir(base_dir)
error("Model outputs directory not found: $base_dir")
end
runs = filter(d -> startswith(d, "run_") && isdir(joinpath(base_dir, d)), readdir(base_dir))
if isempty(runs)
error("No runs found in $base_dir")
end
sort!(runs, rev=true)
latest = joinpath(base_dir, runs[1])
println("Using latest run: $latest")
return latest
end
# =============================================================================
# STEP 1: Load expert_dim.csv from model run data
# =============================================================================
function load_expert_dim(run_dir::String)
expert_dim_file = joinpath(run_dir, "data", "expert_dim.csv")
if !isfile(expert_dim_file)
error("expert_dim.csv not found in $run_dir/data/")
end
expert_dim = CSV.read(expert_dim_file, DataFrame)
println("Loaded expert_dim.csv: $(nrow(expert_dim)) observations")
println(" Unique rr values: $(length(unique(expert_dim.rr_exp_dim)))")
println(" Item indices (var_exp_dim): $(sort(unique(expert_dim.var_exp_dim)))")
println(" Dimensions (dim_idx_exp): $(sort(unique(expert_dim.dim_idx_exp)))")
return expert_dim
end
# =============================================================================
# STEP 2: Selectively load chain columns
# =============================================================================
function load_chains_selective(run_dir::String, needed_rr::Set{Int}, K::Int)
"""Load only the chain columns we need for posterior predictive checks."""
chains_dir = joinpath(run_dir, "chains")
chain_files = sort(filter(f -> endswith(f, ".csv") && startswith(f, "chain_"), readdir(chains_dir)))
if isempty(chain_files)
error("No chain files found in $chains_dir")
end
println("\nLoading $(length(chain_files)) chain files (selective columns)...")
# Build the set of column names we need
needed_cols = Set{String}()
# theta columns: economic_lr.{rr} and galtan.{rr} (cultural cosmopolitan--traditionalist) for each unique rr
for rr in needed_rr
push!(needed_cols, "economic_lr.$rr")
push!(needed_cols, "galtan.$rr")
end
# Item parameters: gamma_exp_intercept.1-K, gamma_exp_slope.1-K
for k in 1:K
push!(needed_cols, "gamma_exp_intercept.$k")
push!(needed_cols, "gamma_exp_slope.$k")
end
# Precision parameter
push!(needed_cols, "phi_exp_dim")
println(" Need $(length(needed_cols)) columns ($(length(needed_rr)) rr × 2 dims + $(2*K) item params + 1 phi)")
# Read the header from first chain to identify column indices
first_chain_path = joinpath(chains_dir, chain_files[1])
header_line = ""
open(first_chain_path) do f
for line in eachline(f)
if !startswith(line, "#")
header_line = line
break
end
end
end
all_cols = split(header_line, ",")
col_indices = Int[]
col_names = String[]
for (i, col) in enumerate(all_cols)
if col in needed_cols
push!(col_indices, i)
push!(col_names, col)
end
end
println(" Found $(length(col_indices))/$(length(needed_cols)) columns in chains")
if length(col_indices) < length(needed_cols)
missing_cols = setdiff(needed_cols, Set(col_names))
n_missing = length(missing_cols)
sample = collect(missing_cols)[1:min(5, n_missing)]
println(" WARNING: Missing columns (showing $( min(5, n_missing))/$n_missing): $sample")
end
# Build a type specification for selective reading
# We'll use CSV.read with select parameter
select_symbols = Symbol.(col_names)
all_chains = DataFrame[]
for (i, cf) in enumerate(chain_files)
path = joinpath(chains_dir, cf)
print(" Loading chain $i: $(cf)... ")
t = @elapsed begin
chain = CSV.read(path, DataFrame; comment="#", select=select_symbols)
end
println("$(nrow(chain)) samples, $(round(t, digits=1))s")
push!(all_chains, chain)
end
combined = vcat(all_chains...)
println("Combined: $(nrow(combined)) total posterior draws")
return combined
end
# =============================================================================
# STEP 3: Compute posterior predictive coverage
# =============================================================================
function compute_posterior_predictive_cic(chains::DataFrame, expert_dim::DataFrame;
ci_level::Float64=0.95,
calibration_levels::Vector{Float64}=[0.50, 0.80, 0.90, 0.95],
seed::Int=42)
"""
Compute posterior predictive coverage for expert dimension observations.
For each expert observation n with observed value y_n:
1. For each posterior draw s:
- Get theta_s = theta[dim, rr] (on logit scale, but chains store inv_logit)
- Compute mu_s = invlogit(gamma_intercept[k] + gamma_slope[k] * logit(theta_s))
- V4 (Beta): Draw y_pred_s ~ Beta(phi * mu_s, phi * (1 - mu_s))
- V5 (Beta-Binomial): Draw y_pred_s ~ Beta(phi * K * mu_s, phi * K * (1 - mu_s))
where K = n_experts for that observation
2. Compute quantile interval of y_pred draws
3. Check if y_n falls within interval
Returns DataFrame with one row per observation plus coverage indicator.
"""
rng = MersenneTwister(seed)
alpha_lower = (1 - ci_level) / 2
alpha_upper = 1 - alpha_lower
N = nrow(expert_dim)
S = nrow(chains) # total posterior draws
# Detect V5 (Beta-Binomial with K-scaling) by presence of n_experts column
has_k_scaling = hasproperty(expert_dim, :n_experts)
if has_k_scaling
k_vec = expert_dim.n_experts
println("\nV5 detected: using Beta(phi*K*mu, phi*K*(1-mu)) with per-observation K")
else
println("\nV4 detected: using Beta(phi*mu, phi*(1-mu))")
end
println("Computing posterior predictive coverage ($(round(Int, 100*ci_level))% level)")
println(" Expert observations: $N")
println(" Posterior draws: $S")
# Pre-extract phi vector
phi_vec = chains[!, :phi_exp_dim]
# Pre-extract gamma vectors for each item k
K = maximum(expert_dim.var_exp_dim)
gamma_int = Dict{Int, Vector{Float64}}()
gamma_slope = Dict{Int, Vector{Float64}}()
for k in 1:K
col_int = Symbol("gamma_exp_intercept.$k")
col_slope = Symbol("gamma_exp_slope.$k")
if hasproperty(chains, col_int) && hasproperty(chains, col_slope)
gamma_int[k] = chains[!, col_int]
gamma_slope[k] = chains[!, col_slope]
end
end
# Allocate result columns
covered = BitVector(undef, N)
pred_lower = Vector{Float64}(undef, N)
pred_upper = Vector{Float64}(undef, N)
pred_median = Vector{Float64}(undef, N)
calibration_covered = Dict(level => BitVector(undef, N) for level in calibration_levels)
# Pre-allocate per-observation draw buffer
y_pred = Vector{Float64}(undef, S)
prog_interval = max(1, N ÷ 20)
for n in 1:N
if n % prog_interval == 0 || n == N
pct = round(100 * n / N, digits=1)
print("\r Progress: $pct% ($n / $N)")
end
rr = expert_dim.rr_exp_dim[n]
dim = expert_dim.dim_idx_exp[n]
k = expert_dim.var_exp_dim[n]
y_obs = expert_dim.val[n]
# Get theta column (chains store inv_logit(theta), i.e. on [0,1] scale)
theta_col = dim == 1 ? Symbol("economic_lr.$rr") : Symbol("galtan.$rr")
if !hasproperty(chains, theta_col) || !haskey(gamma_int, k)
# Missing chain data — mark as not covered
covered[n] = false
pred_lower[n] = NaN
pred_upper[n] = NaN
pred_median[n] = NaN
for level in calibration_levels
calibration_covered[level][n] = false
end
continue
end
theta_star_vec = chains[!, theta_col] # inv_logit(theta), i.e. on [0,1]
g_int = gamma_int[k]
g_slope = gamma_slope[k]
# Effective concentration: phi for V4, phi * n_experts for V5
k_mult = has_k_scaling ? Float64(k_vec[n]) : 1.0
# For each posterior draw, simulate a predictive observation
for s in 1:S
theta_star = theta_star_vec[s]
# Convert back to latent scale for linear predictor
# theta_star is inv_logit(theta), so theta = logit(theta_star)
# Clamp to avoid Inf
theta_star_clamped = clamp(theta_star, 1e-10, 1 - 1e-10)
theta_latent = log(theta_star_clamped / (1 - theta_star_clamped))
# Linear predictor
lin = g_int[s] + g_slope[s] * theta_latent
# Mean of beta
mu = 1 / (1 + exp(-lin))
mu = clamp(mu, 1e-6, 1 - 1e-6)
# Beta parameters: phi * K * mu for V5, phi * mu for V4
phi = phi_vec[s] * k_mult
a = phi * mu
b = phi * (1 - mu)
# Draw from Beta(a, b) via gamma method (no Distributions.jl needed)
y_pred[s] = _rand_beta(rng, a, b)
end
# Compute predictive interval
sort!(y_pred)
idx_lo = max(1, round(Int, alpha_lower * S))
idx_hi = min(S, round(Int, alpha_upper * S))
idx_med = round(Int, 0.5 * S)
pred_lower[n] = y_pred[idx_lo]
pred_upper[n] = y_pred[idx_hi]
pred_median[n] = y_pred[idx_med]
covered[n] = (y_obs >= pred_lower[n]) && (y_obs <= pred_upper[n])
# Compute calibration at several nominal levels from the same predictive
# draws. This avoids repeating the expensive simulation and chain read.
for level in calibration_levels
level_alpha = (1 - level) / 2
level_lo = max(1, round(Int, level_alpha * S))
level_hi = min(S, round(Int, (1 - level_alpha) * S))
calibration_covered[level][n] =
(y_obs >= y_pred[level_lo]) && (y_obs <= y_pred[level_hi])
end
end
println() # newline after progress
# Add results to a copy of expert_dim
result = DataFrame(
rr = expert_dim.rr_exp_dim,
dim_idx = expert_dim.dim_idx_exp,
var_idx = expert_dim.var_exp_dim,
val = expert_dim.val,
party = expert_dim.party,
country = expert_dim.country,
year = expert_dim.year,
project = expert_dim.project,
var = expert_dim.var,
pred_lower = pred_lower,
pred_upper = pred_upper,
pred_median = pred_median,
covered = covered
)
for level in calibration_levels
suffix = lpad(string(round(Int, 100 * level)), 2, '0')
result[!, Symbol("covered_$(suffix)")] = calibration_covered[level]
end
return result
end
function coverage_breakdown(result::DataFrame, group_cols::Vector{Symbol};
covered_col::Symbol=:covered_95)
total_misses = count(.!result[!, covered_col])
grouped = combine(groupby(result, group_cols)) do df
n = nrow(df)
n_covered = count(df[!, covered_col])
n_missed = n - n_covered
p = n_covered / n
ci = wilson_ci(p, n)
DataFrame(
n = n,
covered = n_covered,
missed = n_missed,
observed_coverage = p,
coverage_ci_lower = ci.lower,
coverage_ci_upper = ci.upper,
mean_interval_width = mean(df.pred_upper .- df.pred_lower),
share_of_all_misses = total_misses == 0 ? 0.0 : n_missed / total_misses
)
end
sort!(grouped, [:missed, :n], rev=true)
return grouped
end
function save_detailed_results(result::DataFrame, output_dir::String)
mkpath(output_dir)
result_with_decade = copy(result)
result_with_decade.decade = div.(result_with_decade.year, 10) .* 10
CSV.write(joinpath(output_dir, "posterior_predictive_observations.csv"), result_with_decade)
CSV.write(joinpath(output_dir, "posterior_predictive_by_dimension_source.csv"),
coverage_breakdown(result_with_decade, [:dim_idx, :project]))
CSV.write(joinpath(output_dir, "posterior_predictive_by_dimension_item.csv"),
coverage_breakdown(result_with_decade, [:dim_idx, :project, :var]))
CSV.write(joinpath(output_dir, "posterior_predictive_by_dimension_decade.csv"),
coverage_breakdown(result_with_decade, [:dim_idx, :decade]))
CSV.write(joinpath(output_dir, "posterior_predictive_by_dimension_country.csv"),
coverage_breakdown(result_with_decade, [:dim_idx, :country]))
calibration_rows = NamedTuple[]
for dim in sort(unique(result_with_decade.dim_idx))
subset = filter(:dim_idx => ==(dim), result_with_decade)
for level in (50, 80, 90, 95)
col = Symbol("covered_$(level)")
n = nrow(subset)
n_covered = count(subset[!, col])
push!(calibration_rows, (
dim_idx=dim,
nominal_level=level / 100,
observed_coverage=n_covered / n,
covered=n_covered,
n=n
))
end
end
CSV.write(joinpath(output_dir, "posterior_predictive_calibration_curve.csv"),
DataFrame(calibration_rows))
println("Saved detailed posterior-predictive results to: $output_dir")
end
# =============================================================================
# Beta random variate without Distributions.jl
# =============================================================================
"""
_rand_beta(rng, a, b)
Generate a Beta(a, b) random variate using the Gamma method:
Beta(a,b) = X/(X+Y) where X ~ Gamma(a), Y ~ Gamma(b).
Uses Marsaglia & Tsang (2000) for Gamma generation.
"""
function _rand_beta(rng::AbstractRNG, a::Float64, b::Float64)
x = _rand_gamma(rng, a)
y = _rand_gamma(rng, b)
return x / (x + y)
end
"""
_rand_gamma(rng, shape)
Generate Gamma(shape, 1) random variate using Marsaglia & Tsang (2000).
For shape < 1, uses the rejection method with shape+1 then scales.
"""
function _rand_gamma(rng::AbstractRNG, shape::Float64)
if shape < 1.0
# Gamma(a) = Gamma(a+1) * U^(1/a) where U ~ Uniform(0,1)
return _rand_gamma(rng, shape + 1.0) * rand(rng)^(1.0 / shape)
end
# Marsaglia & Tsang (2000) for shape >= 1
d = shape - 1.0/3.0
c = 1.0 / sqrt(9.0 * d)
while true
local x::Float64
local v::Float64
while true
x = randn(rng)
v = 1.0 + c * x
if v > 0.0
break
end
end
v = v * v * v
u = rand(rng)
if u < 1.0 - 0.0331 * x^2 * x^2
return d * v
end
if log(u) < 0.5 * x^2 + d * (1.0 - v + log(v))
return d * v
end
end
end
# =============================================================================
# STEP 4: Summarize and save results
# =============================================================================
function summarize_coverage(result::DataFrame, ci_level::Float64)
level_pct = round(Int, 100 * ci_level)
println("\n" * "="^60)
println("POSTERIOR PREDICTIVE COVERAGE ($level_pct%)")
println("="^60)
# Overall by dimension
dim_names = Dict(1 => "economic_lr", 2 => "galtan")
display_dim_names = Dict(1 => "economic left-right", 2 => "cultural cosmopolitan--traditionalist")
summary_rows = []
for dim in sort(unique(result.dim_idx))
subset = filter(r -> r.dim_idx == dim, result)
n = nrow(subset)
n_covered = sum(subset.covered)
ppc = n_covered / n
ci = wilson_ci(ppc, n)
dim_name = dim_names[dim]
println(@sprintf("\n %-40s: %.1f%% [%.1f%%, %.1f%%] (%d/%d)",
display_dim_names[dim], 100*ppc, 100*ci.lower, 100*ci.upper, n_covered, n))
push!(summary_rows, (
dimension = dim_name,
cic = ppc,
cic_pct = round(100 * ppc, digits=1),
ci_lower = ci.lower,
ci_upper = ci.upper,
n = n,
covered = n_covered
))
# By project
println("\n By survey source:")
by_project = combine(groupby(subset, :project)) do df
nc = sum(df.covered)
DataFrame(n = nrow(df), covered = nc, cic = nc / nrow(df))
end
sort!(by_project, :n, rev=true)
@printf(" %-12s %6s %8s\n", "Project", "N", "PPC")
for row in eachrow(by_project)
@printf(" %-12s %6d %7.1f%%\n", row.project, row.n, 100*row.cic)
end
# By decade
subset_with_decade = copy(subset)
subset_with_decade.decade = div.(subset_with_decade.year, 10) .* 10
println("\n By decade:")
by_decade = combine(groupby(subset_with_decade, :decade)) do df
nc = sum(df.covered)
DataFrame(n = nrow(df), covered = nc, cic = nc / nrow(df))
end
sort!(by_decade, :decade)
@printf(" %-8s %6s %8s\n", "Decade", "N", "PPC")
for row in eachrow(by_decade)
@printf(" %-8d %6d %7.1f%%\n", row.decade, row.n, 100*row.cic)
end
end
return summary_rows
end
function save_results(result_95::DataFrame, summary_95, summary_80,
by_project_95::Dict, output_dir::String="validation")
if !isdir(output_dir)
mkpath(output_dir)
end
timestamp = Dates.format(now(), "yyyy-mm-dd_HH-MM-SS")
# Summary table (95%)
if !isempty(summary_95)
summary_df = DataFrame(summary_95)
summary_file = joinpath(output_dir, "uncertainty_cic_summary_$timestamp.csv")
CSV.write(summary_file, summary_df)
println("\nSaved: $summary_file")
end
# Also save 80% summary
if !isempty(summary_80)
summary80_df = DataFrame(summary_80)
summary80_file = joinpath(output_dir, "uncertainty_cic_80pct_summary_$timestamp.csv")
CSV.write(summary80_file, summary80_df)
println("Saved: $summary80_file")
end
# By-project tables (95%)
dim_names = Dict(1 => "economic_lr", 2 => "galtan")
for (dim, bp) in by_project_95
project_file = joinpath(output_dir, "uncertainty_$(dim_names[dim])_by_project_$timestamp.csv")
CSV.write(project_file, bp)
println("Saved: $project_file")
end
return summary_95
end
function print_claassen_comparison(summary_95, summary_80)
println("\n" * "="^60)
println("COMPARISON WITH CLAASSEN (2019) BENCHMARKS")
println("="^60)
println("\nClaassen's result:")
println(" CIC (80% CI): 60.3%")
println(" (Using credible intervals for θ, not posterior predictive)")
println()
println("Our results (posterior predictive):")
println()
println("-"^60)
@printf("%-15s %10s %10s %8s\n", "Dimension", "PPC 95%", "PPC 80%", "Status")
println("-"^60)
dim_map_80 = Dict(r.dimension => r for r in summary_80)
for r in summary_95
ppc80 = haskey(dim_map_80, r.dimension) ? dim_map_80[r.dimension].cic : NaN
# Well-calibrated: 95% PPC should be near 95%
status = r.cic >= 0.90 ? "GOOD" : (r.cic >= 0.80 ? "OK" : "LOW")
@printf("%-15s %9.1f%% %9.1f%% %8s\n",
r.dimension, 100*r.cic, 100*ppc80, status)
end
println("-"^60)
println()
println("Interpretation:")
println(" 95% PPC ~95% = well-calibrated uncertainty")
println(" 80% PPC > 60% = exceeds Claassen (2019) benchmark")
end
# =============================================================================
# MAIN
# =============================================================================
function main()
println("="^60)
println("UNCERTAINTY VALIDATION: Posterior Predictive Coverage")
println("="^60)
println("Following Claassen (2019) validation framework")
println("Posterior predictive intervals account for both position")
println("uncertainty AND observation-level measurement noise.")
println()
# Step 0: Parse options and find run directory
run_dir = nothing
output_dir = "validation/revision"
quick_mode = get(ENV, "QUICK_VALIDATION", "0") == "1"
for (i, arg) in enumerate(ARGS)
if arg == "--run-dir" && i < length(ARGS)
run_dir = ARGS[i + 1]
elseif startswith(arg, "--run-dir=")
run_dir = split(arg, "=", limit=2)[2]
elseif arg == "--output-dir" && i < length(ARGS)
output_dir = ARGS[i + 1]
elseif startswith(arg, "--output-dir=")
output_dir = split(arg, "=", limit=2)[2]
elseif arg == "--quick"
quick_mode = true
end
end
if run_dir === nothing
run_dir = find_latest_run()
else
println("Using specified run directory: $run_dir")
end
# Step 1: Load expert_dim.csv
expert_dim = load_expert_dim(run_dir)
# Step 2: Selectively load chains
needed_rr = Set(expert_dim.rr_exp_dim)
K = maximum(expert_dim.var_exp_dim)
chains = load_chains_selective(run_dir, needed_rr, K)
# Step 3a: Compute 95% posterior predictive coverage
result_95 = compute_posterior_predictive_cic(chains, expert_dim; ci_level=0.95)
summary_95 = summarize_coverage(result_95, 0.95)
save_detailed_results(result_95, output_dir)
# Step 3b: Compute 80% posterior predictive coverage (Claassen benchmark)
# Recompute coverage from the same predictive draws but with 80% quantiles
if quick_mode
println("\nQUICK MODE: skipping 80% PPC recomputation")
summary_80 = [(dimension=r.dimension, n=r.n, covered=r.covered, cic=NaN, ci_level=0.80) for r in summary_95]
else
println("\n" * "="^60)
println("RECOMPUTING WITH 80% LEVEL (Claassen comparison)")
println("="^60)
result_80 = compute_posterior_predictive_cic(chains, expert_dim; ci_level=0.80, seed=42)
summary_80 = summarize_coverage(result_80, 0.80)
end
# Build by-project tables for 95%
dim_names = Dict(1 => "economic_lr", 2 => "galtan")
by_project_95 = Dict{Int, DataFrame}()
for dim in sort(unique(result_95.dim_idx))
subset = filter(r -> r.dim_idx == dim, result_95)
bp = combine(groupby(subset, :project)) do df
nc = sum(df.covered)
DataFrame(n = nrow(df), covered = nc, cic = nc / nrow(df))
end
sort!(bp, :n, rev=true)
by_project_95[dim] = bp
end
# Step 4: Save results
save_results(result_95, summary_95, summary_80, by_project_95, output_dir)
# Step 5: Print Claassen comparison
print_claassen_comparison(summary_95, summary_80)
println("\n" * "="^60)
println("VALIDATION COMPLETE")
println("="^60)
return (summary_95=summary_95, summary_80=summary_80)
end
if abspath(PROGRAM_FILE) == @__FILE__
main()
end
+54
View File
@@ -0,0 +1,54 @@
#!/usr/bin/env bash
set -euo pipefail
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd -P)"
project_root="$(cd "$repo_root/.." && pwd -P)"
raw_data_dir="${PARTY2D_RAW_DATA_DIR:-$project_root/_local/raw}"
required_files=(
"poldem/poldem-election_all.csv"
)
optional_files=(
"manifesto/MPDataset_MPDS2025a.csv"
)
echo "Raw data directory: $raw_data_dir"
echo
echo "Required raw inputs for regeneration:"
missing=0
for rel in "${required_files[@]}"; do
path="$raw_data_dir/$rel"
if [ -s "$path" ]; then
bytes="$(wc -c < "$path")"
read -r sha _ < <(sha256sum "$path")
echo " OK $rel ($bytes bytes, sha256=$sha)"
else
echo " MISSING $rel"
missing=1
fi
done
echo
echo "Optional raw inputs used only if cached processed files are regenerated:"
for rel in "${optional_files[@]}"; do
path="$raw_data_dir/$rel"
if [ -s "$path" ]; then
bytes="$(wc -c < "$path")"
read -r sha _ < <(sha256sum "$path")
echo " OK $rel ($bytes bytes, sha256=$sha)"
else
echo " MISSING $rel"
fi
done
if [ "$missing" -ne 0 ]; then
echo
echo "At least one required raw input is missing." >&2
echo "See docs/RAW_DATA_SOURCES.md for download/local-placement instructions." >&2
exit 1
fi
echo
echo "Required raw data preflight passed."
+48
View File
@@ -0,0 +1,48 @@
#!/usr/bin/env bash
set -euo pipefail
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd -P)"
logs_dir="$repo_root/outputs/logs"
model_dir="$repo_root/outputs/model_outputs/latest"
latest_log="$(ls "$logs_dir"/full_run_*.log 2>/dev/null | sort | tail -n 1 || true)"
latest_run="$(ls -d "$model_dir"/run_* 2>/dev/null | sort | tail -n 1 || true)"
echo "party2d full-run progress"
echo "repo: $repo_root"
echo
if [[ -n "$latest_log" ]]; then
echo "Latest durable log: $latest_log"
echo "--- last 40 log lines ---"
tail -n 40 "$latest_log"
else
echo "No durable full_run_*.log found under $logs_dir"
fi
echo
if [[ -n "$latest_run" ]]; then
echo "Latest model run: $latest_run"
metrics="$latest_run/diagnostics/run_metrics.json"
if [[ -f "$metrics" ]]; then
echo "Metrics: $metrics"
echo "Open the JSON metrics file above for run-summary fields."
else
echo "No archived metrics yet. If the run is still sampling, watch the durable log with:"
if [[ -n "$latest_log" ]]; then
echo " tail -f \"$latest_log\""
else
echo " tail -f outputs/logs/full_run_<timestamp>.log"
fi
fi
else
echo "No completed model run found under $model_dir"
fi
echo
echo "To follow a running job live:"
if [[ -n "$latest_log" ]]; then
echo " tail -f \"$latest_log\""
else
echo " tail -f outputs/logs/full_run_<timestamp>.log"
fi
+37
View File
@@ -0,0 +1,37 @@
# Predeclared case-selection rules
These rules were recorded before inspecting the selected parties' position estimates.
## Trajectory figure
The figure will use parties selected for cross-national recognition, family diversity, long election coverage, and relevance to interpreting movement on the two scales. Selection is not based on the magnitude of the estimated movement.
Preselected parties:
- Germany: Social Democratic Party of Germany (PartyFacts 383)
- Germany: Christian Democratic Union (PartyFacts 1375)
- United Kingdom: Labour Party (PartyFacts 1516)
- United Kingdom: Conservative Party (PartyFacts 1567)
- Denmark: Social Democratic Party, short name SD (PartyFacts 379)
- Sweden: Sweden Democrats (PartyFacts 409)
- United States: Democratic Party (PartyFacts 432)
- United States: Republican Party (PartyFacts 809)
The final display may use a subset only to preserve legibility, but exclusions must be based on panel layout or insufficient coverage rather than unattractive results. Both posterior means and 95% latent-position credible intervals will be shown. Text will not infer a cause of movement from the estimates alone.
## Two-dimensional landmarks
Landmarks will be selected from recognizable party-election cases, with representation of economically left/right, culturally cosmopolitan/traditionalist, and moderate combinations. Target years are declared before checking exact positions; if the requested year is absent, the nearest election within three years will be used and disclosed.
- German Social Democratic Party: 1972 and 2021
- German Christian Democratic Union: 1983 and 2021
- UK Labour Party: 1983 and 1997
- UK Conservative Party: 1979 and 2019
- Swedish Social Democratic Labour Party (PartyFacts 487): 1994 and 2022
- Sweden Democrats: 2010 and 2022
- French National Front (PartyFacts 433): 1988 and 2022
- German The Left (PartyFacts 1545): 2021
- US Democratic Party: 2020
- US Republican Party: 2020
Labels may be pruned only to avoid overlap. The plotting-data table will retain all declared cases, including missing cases and substitutions.
+13
View File
@@ -0,0 +1,13 @@
id,topic,input,output,status
1,country coverage,release v0 election-year panel,metadata/country_coverage_v0.csv,complete
2,scale trajectories,release v0 election-year panel,validation/figures/party_trajectories.pdf,complete
3,landmark party space,release v0 election-year panel,validation/figures/party_landmarks.pdf,complete
4,research workflow,documented source and release workflow,validation/figures/research_workflow.pdf,complete
5,party-blocked validation,guarded blocked fit with 82 parties and completed post-estimation extraction,validation/outputs/blocked_validation_summary.csv,complete_with_convergence_limitation
6,predictive coverage,production posterior run run_2026-06-12_09-34-03,validation/outputs/ppc/posterior_predictive_by_dimension_item.csv,complete
7,predictive calibration curve,production posterior run run_2026-06-12_09-34-03,validation/outputs/ppc/posterior_predictive_calibration_curve.csv,complete
8,predictive residual patterns,production posterior predictive observations,validation/outputs/ppc/posterior_predictive_residual_patterns.csv,complete
9,V-Party sensitivity,completed no-V-Party fit ending 2026-06-05_14-57-17,validation/outputs/vparty_sensitivity_groups.csv,complete
10,V-Party subgroup sensitivity,completed no-V-Party fit ending 2026-06-05_14-57-17,validation/outputs/vparty_sensitivity_groups.csv,complete
11,pooled source-support balance,release v0 election-year panel,validation/outputs/source_support_pooled.csv,complete
12,partially pooled source-support heterogeneity,release v0 election-year panel,validation/outputs/source_support_interactions.csv,complete
1 id topic input output status
2 1 country coverage release v0 election-year panel metadata/country_coverage_v0.csv complete
3 2 scale trajectories release v0 election-year panel validation/figures/party_trajectories.pdf complete
4 3 landmark party space release v0 election-year panel validation/figures/party_landmarks.pdf complete
5 4 research workflow documented source and release workflow validation/figures/research_workflow.pdf complete
6 5 party-blocked validation guarded blocked fit with 82 parties and completed post-estimation extraction validation/outputs/blocked_validation_summary.csv complete_with_convergence_limitation
7 6 predictive coverage production posterior run run_2026-06-12_09-34-03 validation/outputs/ppc/posterior_predictive_by_dimension_item.csv complete
8 7 predictive calibration curve production posterior run run_2026-06-12_09-34-03 validation/outputs/ppc/posterior_predictive_calibration_curve.csv complete
9 8 predictive residual patterns production posterior predictive observations validation/outputs/ppc/posterior_predictive_residual_patterns.csv complete
10 9 V-Party sensitivity completed no-V-Party fit ending 2026-06-05_14-57-17 validation/outputs/vparty_sensitivity_groups.csv complete
11 10 V-Party subgroup sensitivity completed no-V-Party fit ending 2026-06-05_14-57-17 validation/outputs/vparty_sensitivity_groups.csv complete
12 11 pooled source-support balance release v0 election-year panel validation/outputs/source_support_pooled.csv complete
13 12 partially pooled source-support heterogeneity release v0 election-year panel validation/outputs/source_support_interactions.csv complete
+30
View File
@@ -0,0 +1,30 @@
# Validation materials
This directory contains reproducible validation code, party-blocked sensitivity analyses, posterior predictive checks, subgroup balance diagnostics, and figure-generation scripts for the party-position panel.
All analyses preserve the production release (`v0`, production run `run_2026-06-12_09-34-03`). Existing posterior draws and completed sensitivity fits are reused wherever possible. The party-blocked expert validation tests prediction for parties whose expert evidence is entirely excluded from estimation.
## Inputs
- `data/releases/party_2d_election_year_panel_v0.csv.gz`
- `data/releases/party_2d_annual_model_output_v0.csv.gz`
- production posterior run `run_2026-06-12_09-34-03`
- completed no-V-Party sensitivity output `party_positions_2026-06-05_14-57-17.csv`
- model-ready input files in `data/`
Large model runs and their chains remain outside Git. This package contains the scripts, manifests, diagnostics, grouped summaries, plotting data, and figures needed to inspect and reproduce the published validation results. Two large row-level intermediate tables used during post-estimation are intentionally excluded; the retained grouped summaries and scripts document all reported calculations.
## Computational notes
The validation fit retains the production warmup length and four-chain design but uses 1,000 retained iterations per chain (4,000 draws total), rather than 2,000 for the production release. The scripts record immutable train/test identifiers and input hashes before fitting and use existing posterior draws for descriptive and posterior-predictive analyses where applicable.
## Reproduction Entry Points
- `run_postprocessing.R`: country coverage, scale illustrations, V-Party sensitivity and partially pooled source-support diagnostics.
- `validate_uncertainty.jl`: production-chain predictive coverage, calibration and source/item/country/decade breakdowns.
- `prepare_blocked_validation.jl`: deterministic party-blocked train/test construction.
- `run_blocked_validation.sh`: guarded low-priority fit with disk, duplicate-process, toolchain and logging checks.
- `monitor_long_run.sh`: non-invasive status, child-process, disk and recent-log monitoring.
- `finalize_blocked_validation.sh`: guarded post-estimation, held-out summaries and compact diagnostic-manifest collection after the fit succeeds.
- `summarize_blocked_validation.jl`: held-out prediction records and grouped performance summaries after the fit.
- `plot_workflow.R`: data-processing workflow schematic.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+68
View File
@@ -0,0 +1,68 @@
#!/usr/bin/env bash
set -euo pipefail
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P)"
cd "$repo_root"
if [[ "${PARTY2D_APPROVE_POSTESTIMATION:-}" != "YES" ]]; then
echo "Refusing to read chains or run post-estimation without explicit approval." >&2
echo "After approval, rerun with PARTY2D_APPROVE_POSTESTIMATION=YES." >&2
exit 64
fi
run_dir="${1:-$repo_root/_local/validation/blocked_party}"
summary_dir="${2:-$repo_root/validation/outputs}"
estimation_dir="$run_dir/estimations"
if [[ ! -f "$run_dir/model_fit.complete" ]]; then
echo "Refusing to finalize: the guarded fit is not marked complete." >&2
exit 1
fi
if [[ ! -f "$run_dir/model_fit.exit_code" ]] || [[ "$(tr -dc '0-9' < "$run_dir/model_fit.exit_code")" != "0" ]]; then
echo "Refusing to finalize: the guarded fit has no successful exit code." >&2
exit 1
fi
mapfile -t model_runs < <(find "$run_dir/model_run/latest" -mindepth 1 -maxdepth 1 -type d -name 'run_*' 2>/dev/null | sort)
if [[ "${#model_runs[@]}" -eq 0 ]]; then
# The production saver currently writes its run under the repository-level
# output directory even when --data-dir points at a validation workspace.
# Match the completed fit by timestamp instead of relying on "latest" alone.
fit_started="$(tr -d '\r\n' < "$run_dir/model_fit.started")"
fit_completed="$(tr -d '\r\n' < "$run_dir/model_fit.complete")"
mapfile -t model_runs < <(find "$repo_root/outputs/model_outputs/latest" -mindepth 1 -maxdepth 1 -type d -name 'run_*' -newermt "$fit_started" ! -newermt "$fit_completed" | sort)
fi
if [[ "${#model_runs[@]}" -ne 1 ]]; then
echo "Expected exactly one blocked model run; found ${#model_runs[@]}." >&2
printf '%s\n' "${model_runs[@]}" >&2
exit 1
fi
model_run="${model_runs[0]}"
mkdir -p "$estimation_dir" "$summary_dir/blocked_fit_diagnostics"
nice -n 10 julia --project=. src/julia/02_post_estimation.jl \
--run-dir "$model_run" --output-dir "$estimation_dir" \
2>&1 | tee "$run_dir/post_estimation.log"
mapfile -t position_files < <(find "$estimation_dir" -maxdepth 1 -type f -name 'party_positions_*.csv' | sort)
if [[ "${#position_files[@]}" -ne 1 ]]; then
echo "Expected exactly one blocked position file; found ${#position_files[@]}." >&2
exit 1
fi
julia --project=. validation/summarize_blocked_validation.jl \
"${position_files[0]}" "$run_dir" "$summary_dir" \
2>&1 | tee "$run_dir/blocked_summary.log"
cp "$model_run/metadata.json" "$summary_dir/blocked_fit_diagnostics/metadata.json"
cp "$model_run/diagnostics/run_metrics.json" "$summary_dir/blocked_fit_diagnostics/run_metrics.json"
cp "$run_dir/blocked_validation_manifest.csv" "$summary_dir/blocked_fit_diagnostics/blocked_validation_manifest.csv"
cp "$run_dir/blocked_validation_strata.csv" "$summary_dir/blocked_fit_diagnostics/blocked_validation_strata.csv"
cp "$run_dir/blocked_parties.csv" "$summary_dir/blocked_fit_diagnostics/blocked_parties.csv"
cp "$run_dir/input_sha256sums.txt" "$summary_dir/blocked_fit_diagnostics/input_sha256sums.txt"
cp "$run_dir/model_fit.environment" "$summary_dir/blocked_fit_diagnostics/model_fit.environment"
cp "$run_dir/stanc_version.txt" "$summary_dir/blocked_fit_diagnostics/stanc_version.txt"
echo "Blocked validation finalized from: $model_run"
echo "Position file: ${position_files[0]}"
echo "Summary directory: $summary_dir"
@@ -0,0 +1,146 @@
============================================================
UNCERTAINTY VALIDATION: Posterior Predictive Coverage
============================================================
Following Claassen (2019) validation framework
Posterior predictive intervals account for both position
uncertainty AND observation-level measurement noise.
Using specified run directory: /projects/party4d/archive/party2d_replication/outputs/model_outputs/latest/run_2026-06-12_09-34-03
Loaded expert_dim.csv: 22994 observations
Unique rr values: 4261
Item indices (var_exp_dim): [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12]
Dimensions (dim_idx_exp): [1, 2]
Loading 4 chain files (selective columns)...
Need 8547 columns (4261 rr × 2 dims + 24 item params + 1 phi)
Found 8547/8547 columns in chains
Loading chain 1: chain_1.csv... 2000 samples, 39.3s
Loading chain 2: chain_2.csv... 2000 samples, 22.4s
Loading chain 3: chain_3.csv... 2000 samples, 24.5s
Loading chain 4: chain_4.csv... 2000 samples, 16.7s
Combined: 8000 total posterior draws
V5 detected: using Beta(phi*K*mu, phi*K*(1-mu)) with per-observation K
Computing posterior predictive coverage (95% level)
Expert observations: 22994
Posterior draws: 8000
Progress: 5.0% (1149 / 22994)
Progress: 10.0% (2298 / 22994)
Progress: 15.0% (3447 / 22994)
Progress: 20.0% (4596 / 22994)
Progress: 25.0% (5745 / 22994)
Progress: 30.0% (6894 / 22994)
Progress: 35.0% (8043 / 22994)
Progress: 40.0% (9192 / 22994)
Progress: 45.0% (10341 / 22994)
Progress: 50.0% (11490 / 22994)
Progress: 55.0% (12639 / 22994)
Progress: 60.0% (13788 / 22994)
Progress: 65.0% (14937 / 22994)
Progress: 70.0% (16086 / 22994)
Progress: 75.0% (17235 / 22994)
Progress: 80.0% (18384 / 22994)
Progress: 84.9% (19533 / 22994)
Progress: 89.9% (20682 / 22994)
Progress: 94.9% (21831 / 22994)
Progress: 99.9% (22980 / 22994)
Progress: 100.0% (22994 / 22994)
============================================================
POSTERIOR PREDICTIVE COVERAGE (95%)
============================================================
economic left-right : 89.8% [89.1%, 90.5%] (6698/7455)
By survey source:
Project N PPC
V-Party 5645 87.5%
CHES 1177 97.3%
POPPA 384 98.7%
GPS 249 94.0%
By decade:
Decade N PPC
1970 658 88.1%
1980 712 88.9%
1990 1454 88.0%
2000 1758 89.4%
2010 2407 90.2%
2020 466 99.1%
cultural cosmopolitan--traditionalist : 84.8% [84.2%, 85.3%] (13170/15539)
By survey source:
Project N PPC
V-Party 14114 83.8%
CHES 1176 94.3%
GPS 249 91.2%
By decade:
Decade N PPC
1970 1633 86.2%
1980 1772 87.9%
1990 3485 85.5%
2000 3965 84.3%
2010 4426 82.1%
2020 258 97.7%
Saved detailed posterior-predictive results to: revision/outputs/ppc
============================================================
RECOMPUTING WITH 80% LEVEL (Claassen comparison)
============================================================
V5 detected: using Beta(phi*K*mu, phi*K*(1-mu)) with per-observation K
Computing posterior predictive coverage (80% level)
Expert observations: 22994
Posterior draws: 8000
Progress: 5.0% (1149 / 22994)
Progress: 10.0% (2298 / 22994)
Progress: 15.0% (3447 / 22994)
Progress: 20.0% (4596 / 22994)
Progress: 25.0% (5745 / 22994)
Progress: 30.0% (6894 / 22994)
Progress: 35.0% (8043 / 22994)
Progress: 40.0% (9192 / 22994)
Progress: 45.0% (10341 / 22994)
Progress: 50.0% (11490 / 22994)
Progress: 55.0% (12639 / 22994)
Progress: 60.0% (13788 / 22994)
Progress: 65.0% (14937 / 22994)
Progress: 70.0% (16086 / 22994)
Progress: 75.0% (17235 / 22994)
Progress: 80.0% (18384 / 22994)
Progress: 84.9% (19533 / 22994)
Progress: 89.9% (20682 / 22994)
Progress: 94.9% (21831 / 22994)
Progress: 99.9% (22980 / 22994)
Progress: 100.0% (22994 / 22994)
============================================================
POSTERIOR PREDICTIVE COVERAGE (80%)
============================================================
economic left-right : 75.9% [74.9%, 76.8%] (5655/7455)
By survey source:
Project N PPC
V-Party 5645 71.5%
CHES 1177 90.1%
POPPA 384 92.2%
GPS 249 82.7%
By decade:
Decade N PPC
1970 658 69.3%
1980 712 73.5%
1990 1454 73.8%
2000 1758 74.1%
2010 2407 77.0%
2020 466 95.7%
cultural cosmopolitan--traditionalist : 67.9% [67.1%, 68.6%] (10544/15539)
By survey source:
Project N PPC
+50
View File
@@ -0,0 +1,50 @@
#!/usr/bin/env bash
set -euo pipefail
run_dir="${1:-_local/validation/blocked_party}"
pid_file="$run_dir/model_fit.pid"
log_file="$run_dir/model_fit.log"
if [[ ! -f "$pid_file" ]]; then
echo "No PID file at $pid_file"
exit 1
fi
pid="$(tr -dc '0-9' < "$pid_file")"
echo "Run directory: $run_dir"
echo "PID: $pid"
date --iso-8601=seconds
if ps -p "$pid" >/dev/null 2>&1; then
echo "Status: running"
ps -o pid,ppid,etimes,%cpu,%mem,rss,vsz,stat,cmd -p "$pid"
echo "Model processes:"
julia_pid="$(pgrep -P "$pid" -f 'julia.*01_run_model[.]jl' | head -1 || true)"
if [[ -n "$julia_pid" ]]; then
process_ids="$pid,$julia_pid"
while IFS= read -r child_pid; do
[[ -n "$child_pid" ]] && process_ids="$process_ids,$child_pid"
done < <(pgrep -P "$julia_pid" || true)
ps -o pid,ppid,etimes,%cpu,%mem,rss,ni,stat,cmd -p "$process_ids"
else
ps -eo pid,ppid,etimes,%cpu,%mem,rss,ni,stat,cmd | awk -v p="$pid" '$2 == p || $1 == p'
fi
else
echo "Status: not running"
fi
echo "Latest sampler progress:"
grep 'Iteration:' "$log_file" 2>/dev/null | tail -12 || true
echo "Recorded sampler events:"
printf ' rejected-proposal exceptions: '
grep -c '^Exception:' "$log_file" 2>/dev/null || true
printf ' divergence messages: '
grep -ci 'divergent transition' "$log_file" 2>/dev/null || true
echo "Disk use:"
du -sh "$run_dir" _local/tmp/blocked_validation 2>/dev/null || true
df -h "$run_dir" | tail -1
echo "Recent log:"
tail -40 "$log_file" 2>/dev/null || true
@@ -0,0 +1,83 @@
party_id,country,region,first_expert_year,last_expert_year,expert_period,text_years,expert_rows,lr_rows,stratum,selected
42,HU,Europe,2010,2024,2010s,3,33,6,Europe / 2010s,true
96,SI,Europe,1992,2019,1990s,7,31,4,Europe / 1990s,true
172,NL,Europe,1999,1999,1990s,5,2,1,Europe / 1990s,true
216,MX,Latin America,1991,2020,1990s,8,74,1,Latin America / 1990s,true
232,CA,North America,1972,2000,pre-1990,18,63,0,North America / pre-1990,true
237,LT,Europe,2004,2019,2000s,4,36,4,Europe / 2000s,true
281,BE,Europe,1971,1974,pre-1990,6,14,2,Europe / pre-1990,true
284,PT,Europe,1987,2024,pre-1990,9,88,9,Europe / pre-1990,true
292,BA,Europe,2002,2019,2000s,6,37,0,Europe / 2000s,true
298,NL,Europe,2006,2024,2000s,5,42,7,Europe / 2000s,true
306,TR,Europe,2002,2024,2000s,5,32,1,Europe / 2000s,true
363,IS,Europe,1971,2024,pre-1990,23,103,9,Europe / pre-1990,true
441,ES,Europe,1977,2024,pre-1990,15,116,9,Europe / pre-1990,true
447,IL,Other,2021,2022,2020s,4,4,2,Other / 2020s,true
466,CZ,Europe,1992,2024,1990s,8,72,8,Europe / 1990s,true
472,SI,Europe,1990,2024,1990s,9,72,8,Europe / 1990s,true
487,SE,Europe,1970,2024,pre-1990,24,123,18,Europe / pre-1990,true
500,BE,Europe,1978,2024,pre-1990,12,102,9,Europe / pre-1990,true
537,BA,Europe,1998,2019,1990s,8,51,0,Europe / 1990s,true
557,IL,Other,1999,2003,1990s,3,14,0,Other / 1990s,true
633,BE,Europe,1971,2024,pre-1990,16,114,9,Europe / pre-1990,true
705,NO,Europe,1973,2024,pre-1990,19,90,11,Europe / pre-1990,true
714,NL,Europe,2014,2023,2010s,3,8,4,Europe / 2010s,true
848,ES,Europe,1999,2024,1990s,15,56,8,Europe / 1990s,true
910,HU,Europe,1990,2010,1990s,5,41,3,Europe / 1990s,true
934,IT,Europe,1972,1992,pre-1990,14,42,7,Europe / pre-1990,true
946,ES,Europe,1999,1999,1990s,6,2,1,Europe / 1990s,true
950,EC,Latin America,1979,2019,pre-1990,4,100,0,Latin America / pre-1990,true
953,IL,Other,2003,2022,2000s,4,18,2,Other / 2000s,true
986,GB,Europe,1999,2024,1990s,7,25,9,Europe / 1990s,true
1004,CA,North America,2004,2023,2000s,7,46,1,North America / 2000s,true
1036,IL,Other,1973,2022,pre-1990,17,104,2,Other / pre-1990,true
1055,IE,Europe,1973,2024,pre-1990,20,102,9,Europe / pre-1990,true
1060,TR,Europe,1973,2024,pre-1990,15,67,1,Europe / pre-1990,true
1072,NO,Europe,1973,2024,pre-1990,19,88,11,Europe / pre-1990,true
1096,FI,Europe,1970,1987,pre-1990,13,42,9,Europe / pre-1990,true
1123,CH,Europe,2023,2024,2020s,13,3,2,Europe / 2020s,true
1126,IT,Europe,1972,2006,pre-1990,10,11,7,Europe / pre-1990,true
1138,LU,Europe,1984,2023,pre-1990,7,48,3,Europe / pre-1990,true
1157,NL,Europe,1977,2024,pre-1990,14,109,9,Europe / pre-1990,true
1166,BA,Europe,1996,2014,1990s,10,49,0,Europe / 1990s,true
1219,ZA,Other,1994,2019,1990s,6,44,0,Other / 1990s,true
1241,MX,Latin America,2019,2020,2010s,7,4,1,Latin America / 2010s,true
1242,SK,Europe,1994,1998,1990s,4,14,0,Europe / 1990s,true
1249,IS,Europe,1971,1995,pre-1990,15,50,8,Europe / pre-1990,true
1274,SE,Europe,1970,2024,pre-1990,24,114,18,Europe / pre-1990,true
1331,MX,Latin America,2006,2020,2000s,5,25,1,Latin America / 2000s,true
1359,PT,Europe,1975,2024,pre-1990,18,114,9,Europe / pre-1990,true
1386,SK,Europe,2010,2024,2010s,3,33,6,Europe / 2010s,true
1388,GB,Europe,1999,2024,1990s,9,32,9,Europe / 1990s,true
1415,CH,Europe,2011,2011,2010s,3,7,0,Europe / 2010s,true
1428,CA,North America,1993,2023,1990s,10,60,1,North America / 1990s,true
1431,HR,Europe,1992,2024,1990s,9,59,5,Europe / 1990s,true
1450,EC,Latin America,1984,2009,pre-1990,3,84,0,Latin America / pre-1990,true
1454,BA,Europe,1996,2019,1990s,9,51,0,Europe / 1990s,true
1459,NL,Europe,2002,2024,2000s,7,16,8,Europe / 2000s,true
1467,NL,Europe,2010,2024,2010s,5,12,6,Europe / 2010s,true
1508,MK,Europe,1998,2019,1990s,9,44,0,Europe / 1990s,true
1540,AU,Asia-Pacific,1972,1972,pre-1990,10,7,0,Asia-Pacific / pre-1990,true
1567,GB,Europe,1970,2024,pre-1990,22,109,9,Europe / pre-1990,true
1665,BG,Europe,1991,2024,1990s,8,75,6,Europe / 1990s,true
1673,BA,Europe,2000,2019,2000s,5,16,0,Europe / 2000s,true
1691,HU,Europe,1990,2024,1990s,9,70,8,Europe / 1990s,true
1715,RO,Europe,1990,2014,1990s,7,43,4,Europe / 1990s,true
1759,CH,Europe,2011,2024,2010s,4,18,3,Europe / 2010s,true
1804,JP,Asia-Pacific,1996,2014,1990s,7,49,0,Asia-Pacific / 1990s,true
1808,CH,Europe,1971,2024,pre-1990,19,95,1,Europe / pre-1990,true
1824,NZ,Asia-Pacific,1972,2019,pre-1990,26,114,0,Asia-Pacific / pre-1990,true
2159,GE,Europe,1999,1999,1990s,3,7,0,Europe / 1990s,true
2168,GE,Europe,1995,2003,1990s,3,21,0,Europe / 1990s,true
2203,RS,Europe,2000,2000,2000s,6,7,0,Europe / 2000s,true
2211,UA,Europe,1998,2006,1990s,5,21,0,Europe / 1990s,true
2228,UA,Europe,2002,2012,2000s,3,28,0,Europe / 2000s,true
2235,RU,Europe,1993,1993,1990s,3,7,0,Europe / 1990s,true
2256,RU,Europe,2003,2019,2000s,3,30,0,Europe / 2000s,true
2307,KR,Asia-Pacific,2000,2012,2000s,5,28,0,Asia-Pacific / 2000s,true
3162,ME,Europe,1998,2019,1990s,8,51,0,Europe / 1990s,true
3171,BA,Europe,2010,2019,2010s,3,23,0,Europe / 2010s,true
3185,ME,Europe,1998,2016,1990s,5,49,0,Europe / 1990s,true
3916,CO,Latin America,2020,2020,2020s,4,2,1,Latin America / 2020s,true
3955,IL,Other,2019,2019,2010s,4,7,0,Other / 2010s,true
5453,AU,Asia-Pacific,2019,2019,2010s,3,2,0,Asia-Pacific / 2010s,true
1 party_id country region first_expert_year last_expert_year expert_period text_years expert_rows lr_rows stratum selected
2 42 HU Europe 2010 2024 2010s 3 33 6 Europe / 2010s true
3 96 SI Europe 1992 2019 1990s 7 31 4 Europe / 1990s true
4 172 NL Europe 1999 1999 1990s 5 2 1 Europe / 1990s true
5 216 MX Latin America 1991 2020 1990s 8 74 1 Latin America / 1990s true
6 232 CA North America 1972 2000 pre-1990 18 63 0 North America / pre-1990 true
7 237 LT Europe 2004 2019 2000s 4 36 4 Europe / 2000s true
8 281 BE Europe 1971 1974 pre-1990 6 14 2 Europe / pre-1990 true
9 284 PT Europe 1987 2024 pre-1990 9 88 9 Europe / pre-1990 true
10 292 BA Europe 2002 2019 2000s 6 37 0 Europe / 2000s true
11 298 NL Europe 2006 2024 2000s 5 42 7 Europe / 2000s true
12 306 TR Europe 2002 2024 2000s 5 32 1 Europe / 2000s true
13 363 IS Europe 1971 2024 pre-1990 23 103 9 Europe / pre-1990 true
14 441 ES Europe 1977 2024 pre-1990 15 116 9 Europe / pre-1990 true
15 447 IL Other 2021 2022 2020s 4 4 2 Other / 2020s true
16 466 CZ Europe 1992 2024 1990s 8 72 8 Europe / 1990s true
17 472 SI Europe 1990 2024 1990s 9 72 8 Europe / 1990s true
18 487 SE Europe 1970 2024 pre-1990 24 123 18 Europe / pre-1990 true
19 500 BE Europe 1978 2024 pre-1990 12 102 9 Europe / pre-1990 true
20 537 BA Europe 1998 2019 1990s 8 51 0 Europe / 1990s true
21 557 IL Other 1999 2003 1990s 3 14 0 Other / 1990s true
22 633 BE Europe 1971 2024 pre-1990 16 114 9 Europe / pre-1990 true
23 705 NO Europe 1973 2024 pre-1990 19 90 11 Europe / pre-1990 true
24 714 NL Europe 2014 2023 2010s 3 8 4 Europe / 2010s true
25 848 ES Europe 1999 2024 1990s 15 56 8 Europe / 1990s true
26 910 HU Europe 1990 2010 1990s 5 41 3 Europe / 1990s true
27 934 IT Europe 1972 1992 pre-1990 14 42 7 Europe / pre-1990 true
28 946 ES Europe 1999 1999 1990s 6 2 1 Europe / 1990s true
29 950 EC Latin America 1979 2019 pre-1990 4 100 0 Latin America / pre-1990 true
30 953 IL Other 2003 2022 2000s 4 18 2 Other / 2000s true
31 986 GB Europe 1999 2024 1990s 7 25 9 Europe / 1990s true
32 1004 CA North America 2004 2023 2000s 7 46 1 North America / 2000s true
33 1036 IL Other 1973 2022 pre-1990 17 104 2 Other / pre-1990 true
34 1055 IE Europe 1973 2024 pre-1990 20 102 9 Europe / pre-1990 true
35 1060 TR Europe 1973 2024 pre-1990 15 67 1 Europe / pre-1990 true
36 1072 NO Europe 1973 2024 pre-1990 19 88 11 Europe / pre-1990 true
37 1096 FI Europe 1970 1987 pre-1990 13 42 9 Europe / pre-1990 true
38 1123 CH Europe 2023 2024 2020s 13 3 2 Europe / 2020s true
39 1126 IT Europe 1972 2006 pre-1990 10 11 7 Europe / pre-1990 true
40 1138 LU Europe 1984 2023 pre-1990 7 48 3 Europe / pre-1990 true
41 1157 NL Europe 1977 2024 pre-1990 14 109 9 Europe / pre-1990 true
42 1166 BA Europe 1996 2014 1990s 10 49 0 Europe / 1990s true
43 1219 ZA Other 1994 2019 1990s 6 44 0 Other / 1990s true
44 1241 MX Latin America 2019 2020 2010s 7 4 1 Latin America / 2010s true
45 1242 SK Europe 1994 1998 1990s 4 14 0 Europe / 1990s true
46 1249 IS Europe 1971 1995 pre-1990 15 50 8 Europe / pre-1990 true
47 1274 SE Europe 1970 2024 pre-1990 24 114 18 Europe / pre-1990 true
48 1331 MX Latin America 2006 2020 2000s 5 25 1 Latin America / 2000s true
49 1359 PT Europe 1975 2024 pre-1990 18 114 9 Europe / pre-1990 true
50 1386 SK Europe 2010 2024 2010s 3 33 6 Europe / 2010s true
51 1388 GB Europe 1999 2024 1990s 9 32 9 Europe / 1990s true
52 1415 CH Europe 2011 2011 2010s 3 7 0 Europe / 2010s true
53 1428 CA North America 1993 2023 1990s 10 60 1 North America / 1990s true
54 1431 HR Europe 1992 2024 1990s 9 59 5 Europe / 1990s true
55 1450 EC Latin America 1984 2009 pre-1990 3 84 0 Latin America / pre-1990 true
56 1454 BA Europe 1996 2019 1990s 9 51 0 Europe / 1990s true
57 1459 NL Europe 2002 2024 2000s 7 16 8 Europe / 2000s true
58 1467 NL Europe 2010 2024 2010s 5 12 6 Europe / 2010s true
59 1508 MK Europe 1998 2019 1990s 9 44 0 Europe / 1990s true
60 1540 AU Asia-Pacific 1972 1972 pre-1990 10 7 0 Asia-Pacific / pre-1990 true
61 1567 GB Europe 1970 2024 pre-1990 22 109 9 Europe / pre-1990 true
62 1665 BG Europe 1991 2024 1990s 8 75 6 Europe / 1990s true
63 1673 BA Europe 2000 2019 2000s 5 16 0 Europe / 2000s true
64 1691 HU Europe 1990 2024 1990s 9 70 8 Europe / 1990s true
65 1715 RO Europe 1990 2014 1990s 7 43 4 Europe / 1990s true
66 1759 CH Europe 2011 2024 2010s 4 18 3 Europe / 2010s true
67 1804 JP Asia-Pacific 1996 2014 1990s 7 49 0 Asia-Pacific / 1990s true
68 1808 CH Europe 1971 2024 pre-1990 19 95 1 Europe / pre-1990 true
69 1824 NZ Asia-Pacific 1972 2019 pre-1990 26 114 0 Asia-Pacific / pre-1990 true
70 2159 GE Europe 1999 1999 1990s 3 7 0 Europe / 1990s true
71 2168 GE Europe 1995 2003 1990s 3 21 0 Europe / 1990s true
72 2203 RS Europe 2000 2000 2000s 6 7 0 Europe / 2000s true
73 2211 UA Europe 1998 2006 1990s 5 21 0 Europe / 1990s true
74 2228 UA Europe 2002 2012 2000s 3 28 0 Europe / 2000s true
75 2235 RU Europe 1993 1993 1990s 3 7 0 Europe / 1990s true
76 2256 RU Europe 2003 2019 2000s 3 30 0 Europe / 2000s true
77 2307 KR Asia-Pacific 2000 2012 2000s 5 28 0 Asia-Pacific / 2000s true
78 3162 ME Europe 1998 2019 1990s 8 51 0 Europe / 1990s true
79 3171 BA Europe 2010 2019 2010s 3 23 0 Europe / 2010s true
80 3185 ME Europe 1998 2016 1990s 5 49 0 Europe / 1990s true
81 3916 CO Latin America 2020 2020 2020s 4 2 1 Latin America / 2020s true
82 3955 IL Other 2019 2019 2010s 4 7 0 Other / 2010s true
83 5453 AU Asia-Pacific 2019 2019 2010s 3 2 0 Asia-Pacific / 2010s true
@@ -0,0 +1,15 @@
field,value
created_at,2026-08-12T17:24:30.781
seed,20260812
holdout_share,0.2
eligible_parties,388
blocked_parties,82
text_rows_train,38202
expert_rows_train,21162
expert_rows_test,3916
lr_rows_train,1897
lr_rows_test,310
text_sha256,8ba73dc9d378037333cc4ef1629d59350237f911b4fcd3dfd319cefb4769e23a
expert_full_sha256,30b53244055c18ae480197cfde4cdd86f30e0c9bd5af60d371caa5647b2a362a
lr_full_sha256,66912cc39d63b4bf63506220268e2eee2d5d94fc9c8b861eb0e25d7e79f492d2
union_mapping_sha256,3861d65e279458e339a220df6040c98f0f5237476065d01c855e38ad2164fe5c
1 field value
2 created_at 2026-08-12T17:24:30.781
3 seed 20260812
4 holdout_share 0.2
5 eligible_parties 388
6 blocked_parties 82
7 text_rows_train 38202
8 expert_rows_train 21162
9 expert_rows_test 3916
10 lr_rows_train 1897
11 lr_rows_test 310
12 text_sha256 8ba73dc9d378037333cc4ef1629d59350237f911b4fcd3dfd319cefb4769e23a
13 expert_full_sha256 30b53244055c18ae480197cfde4cdd86f30e0c9bd5af60d371caa5647b2a362a
14 lr_full_sha256 66912cc39d63b4bf63506220268e2eee2d5d94fc9c8b861eb0e25d7e79f492d2
15 union_mapping_sha256 3861d65e279458e339a220df6040c98f0f5237476065d01c855e38ad2164fe5c
@@ -0,0 +1,23 @@
stratum,eligible_parties,blocked_parties
Asia-Pacific / pre-1990,12,2
Europe / 1990s,121,24
Europe / pre-1990,103,21
Europe / 2010s,34,7
Europe / 2000s,47,9
North America / pre-1990,6,1
Latin America / pre-1990,10,2
Latin America / 1990s,2,1
Latin America / 2000s,7,1
Other / 2000s,6,1
Asia-Pacific / 2010s,4,1
Other / 2020s,2,1
Other / 1990s,9,2
Asia-Pacific / 1990s,5,1
Other / pre-1990,3,1
Europe / 2020s,3,1
North America / 2000s,2,1
Asia-Pacific / 2000s,4,1
Other / 2010s,5,1
Latin America / 2010s,1,1
North America / 1990s,1,1
Latin America / 2020s,1,1
1 stratum eligible_parties blocked_parties
2 Asia-Pacific / pre-1990 12 2
3 Europe / 1990s 121 24
4 Europe / pre-1990 103 21
5 Europe / 2010s 34 7
6 Europe / 2000s 47 9
7 North America / pre-1990 6 1
8 Latin America / pre-1990 10 2
9 Latin America / 1990s 2 1
10 Latin America / 2000s 7 1
11 Other / 2000s 6 1
12 Asia-Pacific / 2010s 4 1
13 Other / 2020s 2 1
14 Other / 1990s 9 2
15 Asia-Pacific / 1990s 5 1
16 Other / pre-1990 3 1
17 Europe / 2020s 3 1
18 North America / 2000s 2 1
19 Asia-Pacific / 2000s 4 1
20 Other / 2010s 5 1
21 Latin America / 2010s 1 1
22 North America / 1990s 1 1
23 Latin America / 2020s 1 1
@@ -0,0 +1,5 @@
8ba73dc9d378037333cc4ef1629d59350237f911b4fcd3dfd319cefb4769e23a _local/revision/blocked_party/text_data.csv
a1b5031df7754ba33234e8e2ec40bb0f0d10b0aed12b277a8ea1f7876fb0f658 _local/revision/blocked_party/expert.csv
8129e85cd13877c7a5c8698430be279265d1bb9b9dd16438f55b7159597b5af7 _local/revision/blocked_party/lr_data.csv
ed05b52f16139ebce88d3d281e19f4b8140fa1b0f0791fa8fb00ac2ccf347c1f _local/revision/blocked_party/expert_test.csv
fcf131031e9bac5231fdd829b4d2f0b6ac8d9a2de748175a5fb897be048a7abe _local/revision/blocked_party/lr_data_test.csv
@@ -0,0 +1,39 @@
{
"year0": 1943,
"mean_ess": 3900.832212330333,
"num_samples": 1000,
"max_depth": 15,
"num_chains": 4,
"run_id": "run_2026-08-13_00-02-50",
"files": {
"chain_size_gb": 2.28,
"data": [
"expert_dim.csv",
"expert_lr.csv",
"segment_info.csv",
"segment_year_map.csv",
"stan_data.json",
"text_data.csv"
],
"total_size_gb": 9.12,
"chains": [
"chains/chain_1.csv",
"chains/chain_2.csv",
"chains/chain_3.csv",
"chains/chain_4.csv"
]
},
"max_rhat": 1.02529,
"model_file": "models/stan_model_2dim_v6.stan",
"dimensions": [
"economic_lr",
"galtan"
],
"convergence_status": "excellent",
"mean_rhat": 1.001168748932864,
"model_version": "2dim",
"min_ess": 136.47,
"num_warmup": 1000,
"timestamp": "2026-08-13_00-02-50",
"adapt_delta": 0.95
}
@@ -0,0 +1,6 @@
CMDSTAN_HOME=/opt/agent-tools/cmdstan-2.39.0
JULIA_NUM_THREADS=4
STAN_NUM_THREADS=1
PARTY2D_NUM_CHAINS=4
PARTY2D_NUM_WARMUP=1000
PARTY2D_NUM_SAMPLES=1000
File diff suppressed because one or more lines are too long
@@ -0,0 +1 @@
stanc3 v2.39.0 (Unix)
@@ -0,0 +1,48 @@
dimension,group_type,group,n,parties,pearson_r,mae,rmse,bias_expert_minus_model,latent_interval_overlap_95,latent_interval_overlap_ci_lower,latent_interval_overlap_ci_upper,mean_interval_width,mean_nearest_text_distance
economic_lr,overall,All matched held-out ratings,1095,76,0.5475349609862943,0.15659217513079526,0.20393539783060147,-0.05532600480562651,0.5168949771689497,0.4872890521846218,0.5463827709425695,0.26430020129022835,0.16712328767123288
economic_lr,decade,1970s,118,23,0.6073768641760265,0.14287929777127256,0.19854739676618133,-0.10205433290640958,0.6101694915254238,0.5200255656951949,0.6933662479382441,0.25169848501186437,0.0
economic_lr,decade,2010s,354,57,0.5992779154544512,0.1501300890725984,0.1885321056791222,0.021169532897549623,0.5254237288135594,0.47341101610110464,0.57689056985201,0.26575748477902544,0.2740112994350282
economic_lr,decade,2000s,295,55,0.5565144172854626,0.15858892901808844,0.20126848038834502,-0.05233537589313869,0.4847457627118644,0.4282780233542677,0.5416056876150218,0.26056229448652546,0.17288135593220338
economic_lr,decade,1990s,208,50,0.6126480905939616,0.16338137608018816,0.21895575578982457,-0.125792230775093,0.5048076923076923,0.43739172918853736,0.5720492871179859,0.27198704433737986,0.15384615384615385
economic_lr,decade,1980s,108,20,0.4965786440080896,0.17854227265283956,0.23865365933045032,-0.1349662304081571,0.4722222222222222,0.38064769707967805,0.56570500173722,0.24913035199722225,0.0
economic_lr,decade,2020s,12,7,0.7503511986609598,0.11774978066574382,0.1393618688955034,0.0122078457365813,0.75,0.46768966087934005,0.9110599603710386,0.4404074548520834,0.25
economic_lr,region,Asia-Pacific,58,5,0.423328443816105,0.17092882695033917,0.20862210989480942,-0.0891987769436264,0.39655172413793105,0.28088568950111137,0.5250701719255011,0.24057815327112078,0.017241379310344827
economic_lr,region,Europe,893,59,0.5432249309980023,0.1511392922038456,0.20048220769183805,-0.04801670847392163,0.5397536394176932,0.5069623693775478,0.5722043418900828,0.2582198076309352,0.17245240761478164
economic_lr,region,North America,48,3,0.2266003540841267,0.19345335195896976,0.23606823520460213,-0.0729662252491865,0.3333333333333333,0.2167660137089678,0.47460153667527943,0.24118537241145843,0.0
economic_lr,region,Latin America,44,4,0.5475305121725508,0.233854863071928,0.26544986189092146,-0.22241677738442417,0.3409090909090909,0.21875632628639946,0.4886113202806049,0.3793346702613636,0.6363636363636364
economic_lr,region,Other,52,5,0.7564368993112075,0.13484224995906788,0.16103703074927944,0.014599836243402173,0.5769230769230769,0.4419408497943149,0.7013215209113954,0.3191783834884615,0.0
economic_lr,text_distance_class,direct text,971,74,0.5470435275586387,0.15948847158646431,0.20840482347744635,-0.06628643977796411,0.505664263645726,0.47425628054953844,0.537027603929745,0.2590549584609939,0.0
economic_lr,text_distance_class,nearby text (1--3 years),124,41,0.6274910136969545,0.1339123053045472,0.16479794566578806,0.030501272276146137,0.6048387096774194,0.5168822918241138,0.686494386815703,0.3053738366707661,1.4758064516129032
economic_lr,project,V-Party,914,70,0.5307923250620764,0.1655176264187938,0.21441603701361558,-0.07700198607575016,0.4890590809628009,0.45676497600150245,0.5214447717371051,0.2606530705033918,0.0437636761487965
economic_lr,project,GPS,26,26,0.6972897756799338,0.13173814981538515,0.16398882896345623,0.04775018790861701,0.5,0.32060306315002074,0.6793969368499793,0.28959384095,0.8076923076923077
economic_lr,project,CHES,129,31,0.8041829585092515,0.10700941979056187,0.13157526424371663,0.044641159665681336,0.6821705426356589,0.5975442490042815,0.7562605832169504,0.2815578695228682,0.7829457364341085
economic_lr,project,POPPA,26,21,0.8377220415965767,0.11368900666387213,0.15035280874886425,0.10760098186837214,0.6923076923076923,0.5001138848576876,0.8349887906020732,0.2815926515211539,0.8076923076923077
economic_lr,item,lrecon_vparty,457,70,0.6526133931659629,0.12080138892236934,0.1567424568814474,-0.02190935296123517,0.6148796498905909,0.5694819726873812,0.6583620413946809,0.2606530705033917,0.0437636761487965
economic_lr,item,welf_vparty,457,70,0.4769148611627237,0.21023386391521848,0.2595771100617617,-0.13209461919026505,0.36323851203501095,0.32045365106651325,0.40830347502626996,0.2606530705033917,0.0437636761487965
economic_lr,item,lrecon_gps,26,26,0.6972897756799338,0.13173814981538515,0.16398882896345623,0.04775018790861701,0.5,0.32060306315002074,0.6793969368499793,0.28959384095,0.8076923076923077
economic_lr,item,lrecon_ches,129,31,0.8041829585092515,0.10700941979056187,0.13157526424371663,0.044641159665681336,0.6821705426356589,0.5975442490042815,0.7562605832169504,0.2815578695228682,0.7829457364341085
economic_lr,item,lrecon_poppa,26,21,0.8377220415965767,0.11368900666387213,0.15035280874886425,0.10760098186837214,0.6923076923076923,0.5001138848576876,0.8349887906020732,0.2815926515211539,0.8076923076923077
galtan,overall,All matched held-out ratings,2428,76,0.3879064541660041,0.18914006264912334,0.23816373704876254,-0.013710351768187365,0.4126853377265239,0.39325537210953215,0.4323911667110884,0.258665849890383,0.0914332784184514
galtan,decade,1970s,287,23,0.2402564326635899,0.21665443860032066,0.25724972613838026,0.0744907055227787,0.3344947735191638,0.2824119799338199,0.3909497399852015,0.26474203940810126,0.0
galtan,decade,2010s,693,57,0.4606647139522965,0.1935411541181932,0.24590414523794452,-0.050660741407812654,0.3838383838383838,0.3483644977339694,0.4205930386702865,0.24824147265028856,0.12265512265512266
galtan,decade,2000s,676,55,0.3870265819854018,0.18629380051270433,0.23690167706252432,-0.0468645901306479,0.4275147928994083,0.39073352449867466,0.4651152496863708,0.258417938360392,0.11538461538461539
galtan,decade,1990s,499,50,0.41212741006828396,0.17126387793509984,0.2212178226996366,-0.019602997434269034,0.48897795591182364,0.4453697785277533,0.5327545453149837,0.26606359520380773,0.11823647294589178
galtan,decade,1980s,266,20,0.3476988123327735,0.18921982830486436,0.2308346130602219,0.08154082475673356,0.37969924812030076,0.3234807111340447,0.4393431084697526,0.2599239549642858,0.0
galtan,decade,2020s,7,5,0.6324675234649824,0.17149569493926073,0.20386312323528286,0.030403782877832106,0.8571428571428571,0.48686549668097007,0.9743210440510253,0.49033504546428575,0.0
galtan,region,Asia-Pacific,142,5,-0.06990999699587813,0.2028535844559974,0.2421448847066388,-0.04764709776122232,0.43661971830985913,0.3577770958258817,0.5188013289856945,0.31393695714260567,0.007042253521126761
galtan,region,Europe,1938,59,0.4006536844413476,0.18821986618869885,0.2391185533726696,-0.007048457123688323,0.4107327141382869,0.3890267037664939,0.4327919244882515,0.24984948892156866,0.07791537667698659
galtan,region,North America,117,3,0.15158098770839884,0.2020264962636105,0.24484307272242728,0.04760720430768876,0.2905982905982906,0.21602760266916315,0.37848289702630594,0.23609932028311967,0.0
galtan,region,Latin America,110,4,0.510877747351673,0.15390455215418858,0.19797883572628092,-0.07917475272712499,0.6272727272727273,0.5340503169531756,0.7119054672241376,0.3662500658409092,0.6363636363636364
galtan,region,Other,121,5,0.3370413394210674,0.20735670781668103,0.24493355112293227,-0.08036162321796078,0.33884297520661155,0.2606254540535898,0.4269786779024248,0.25902643284276866,0.0
galtan,text_distance_class,direct text,2293,74,0.3817373914639877,0.1904499145514716,0.23993647489228756,-0.014524078018235874,0.404709986916703,0.38479506099767413,0.42494366891735347,0.25436052910535323,0.0
galtan,text_distance_class,nearby text (1--3 years),135,40,0.4399561874640988,0.16689198552257195,0.20573340547684066,0.00011093927893245328,0.5481481481481482,0.46402181812792126,0.6296100616545071,0.3317925207057409,1.6444444444444444
galtan,project,V-Party,2273,70,0.3740693791970041,0.1911074338064693,0.24060334153635682,-0.018396438536216624,0.4047514298284206,0.3847495239974894,0.4250747518765987,0.2566477345279916,0.04399472063352398
galtan,project,GPS,26,26,0.5443693144157825,0.17671657291477977,0.21780593561007713,0.012007051328791764,0.46153846153846156,0.28755582695234905,0.6454236379556988,0.2777555471798077,0.8076923076923077
galtan,project,CHES,129,31,0.5869373239223193,0.15697863700916778,0.19496790336712624,0.06367587104738649,0.5426356589147286,0.456675986048851,0.6261294002156924,0.2903778195740311,0.7829457364341085
galtan,item,culsup_vparty,457,70,0.5366188718286774,0.16311644869435024,0.21359963974416982,0.011168223905568687,0.4638949671772429,0.418663126356892,0.5097287549316027,0.25691081806515326,0.0437636761487965
galtan,item,gender_vparty,445,70,0.32459509946925547,0.2164337137525867,0.2608755427368426,0.12306040878157333,0.34606741573033706,0.3033546905138159,0.39141513474302303,0.25556702282926974,0.0449438202247191
galtan,item,immig_vparty,457,70,0.5004863614572931,0.13311846254318913,0.16732357853899973,-0.011445561652418225,0.5776805251641138,0.5319316389137728,0.6221343134655263,0.25691081806515326,0.0437636761487965
galtan,item,lgbt_vparty,457,70,0.46354086621523605,0.15471230984590562,0.18981764808744647,0.04310285191432138,0.474835886214442,0.42945203114163966,0.5206392800594324,0.25691081806515326,0.0437636761487965
galtan,item,relig_vparty,457,70,0.4185372751933906,0.288821256864484,0.33467597534992094,-0.25415371263710096,0.15973741794310722,0.12900420700742332,0.19614352271142144,0.25691081806515326,0.0437636761487965
galtan,item,libcon_gps,26,26,0.5443693144157825,0.17671657291477977,0.21780593561007713,0.012007051328791764,0.46153846153846156,0.28755582695234905,0.6454236379556988,0.2777555471798077,0.8076923076923077
galtan,item,galtan_ches,129,31,0.5869373239223193,0.15697863700916778,0.19496790336712624,0.06367587104738649,0.5426356589147286,0.456675986048851,0.6261294002156924,0.2903778195740311,0.7829457364341085
1 dimension group_type group n parties pearson_r mae rmse bias_expert_minus_model latent_interval_overlap_95 latent_interval_overlap_ci_lower latent_interval_overlap_ci_upper mean_interval_width mean_nearest_text_distance
2 economic_lr overall All matched held-out ratings 1095 76 0.5475349609862943 0.15659217513079526 0.20393539783060147 -0.05532600480562651 0.5168949771689497 0.4872890521846218 0.5463827709425695 0.26430020129022835 0.16712328767123288
3 economic_lr decade 1970s 118 23 0.6073768641760265 0.14287929777127256 0.19854739676618133 -0.10205433290640958 0.6101694915254238 0.5200255656951949 0.6933662479382441 0.25169848501186437 0.0
4 economic_lr decade 2010s 354 57 0.5992779154544512 0.1501300890725984 0.1885321056791222 0.021169532897549623 0.5254237288135594 0.47341101610110464 0.57689056985201 0.26575748477902544 0.2740112994350282
5 economic_lr decade 2000s 295 55 0.5565144172854626 0.15858892901808844 0.20126848038834502 -0.05233537589313869 0.4847457627118644 0.4282780233542677 0.5416056876150218 0.26056229448652546 0.17288135593220338
6 economic_lr decade 1990s 208 50 0.6126480905939616 0.16338137608018816 0.21895575578982457 -0.125792230775093 0.5048076923076923 0.43739172918853736 0.5720492871179859 0.27198704433737986 0.15384615384615385
7 economic_lr decade 1980s 108 20 0.4965786440080896 0.17854227265283956 0.23865365933045032 -0.1349662304081571 0.4722222222222222 0.38064769707967805 0.56570500173722 0.24913035199722225 0.0
8 economic_lr decade 2020s 12 7 0.7503511986609598 0.11774978066574382 0.1393618688955034 0.0122078457365813 0.75 0.46768966087934005 0.9110599603710386 0.4404074548520834 0.25
9 economic_lr region Asia-Pacific 58 5 0.423328443816105 0.17092882695033917 0.20862210989480942 -0.0891987769436264 0.39655172413793105 0.28088568950111137 0.5250701719255011 0.24057815327112078 0.017241379310344827
10 economic_lr region Europe 893 59 0.5432249309980023 0.1511392922038456 0.20048220769183805 -0.04801670847392163 0.5397536394176932 0.5069623693775478 0.5722043418900828 0.2582198076309352 0.17245240761478164
11 economic_lr region North America 48 3 0.2266003540841267 0.19345335195896976 0.23606823520460213 -0.0729662252491865 0.3333333333333333 0.2167660137089678 0.47460153667527943 0.24118537241145843 0.0
12 economic_lr region Latin America 44 4 0.5475305121725508 0.233854863071928 0.26544986189092146 -0.22241677738442417 0.3409090909090909 0.21875632628639946 0.4886113202806049 0.3793346702613636 0.6363636363636364
13 economic_lr region Other 52 5 0.7564368993112075 0.13484224995906788 0.16103703074927944 0.014599836243402173 0.5769230769230769 0.4419408497943149 0.7013215209113954 0.3191783834884615 0.0
14 economic_lr text_distance_class direct text 971 74 0.5470435275586387 0.15948847158646431 0.20840482347744635 -0.06628643977796411 0.505664263645726 0.47425628054953844 0.537027603929745 0.2590549584609939 0.0
15 economic_lr text_distance_class nearby text (1--3 years) 124 41 0.6274910136969545 0.1339123053045472 0.16479794566578806 0.030501272276146137 0.6048387096774194 0.5168822918241138 0.686494386815703 0.3053738366707661 1.4758064516129032
16 economic_lr project V-Party 914 70 0.5307923250620764 0.1655176264187938 0.21441603701361558 -0.07700198607575016 0.4890590809628009 0.45676497600150245 0.5214447717371051 0.2606530705033918 0.0437636761487965
17 economic_lr project GPS 26 26 0.6972897756799338 0.13173814981538515 0.16398882896345623 0.04775018790861701 0.5 0.32060306315002074 0.6793969368499793 0.28959384095 0.8076923076923077
18 economic_lr project CHES 129 31 0.8041829585092515 0.10700941979056187 0.13157526424371663 0.044641159665681336 0.6821705426356589 0.5975442490042815 0.7562605832169504 0.2815578695228682 0.7829457364341085
19 economic_lr project POPPA 26 21 0.8377220415965767 0.11368900666387213 0.15035280874886425 0.10760098186837214 0.6923076923076923 0.5001138848576876 0.8349887906020732 0.2815926515211539 0.8076923076923077
20 economic_lr item lrecon_vparty 457 70 0.6526133931659629 0.12080138892236934 0.1567424568814474 -0.02190935296123517 0.6148796498905909 0.5694819726873812 0.6583620413946809 0.2606530705033917 0.0437636761487965
21 economic_lr item welf_vparty 457 70 0.4769148611627237 0.21023386391521848 0.2595771100617617 -0.13209461919026505 0.36323851203501095 0.32045365106651325 0.40830347502626996 0.2606530705033917 0.0437636761487965
22 economic_lr item lrecon_gps 26 26 0.6972897756799338 0.13173814981538515 0.16398882896345623 0.04775018790861701 0.5 0.32060306315002074 0.6793969368499793 0.28959384095 0.8076923076923077
23 economic_lr item lrecon_ches 129 31 0.8041829585092515 0.10700941979056187 0.13157526424371663 0.044641159665681336 0.6821705426356589 0.5975442490042815 0.7562605832169504 0.2815578695228682 0.7829457364341085
24 economic_lr item lrecon_poppa 26 21 0.8377220415965767 0.11368900666387213 0.15035280874886425 0.10760098186837214 0.6923076923076923 0.5001138848576876 0.8349887906020732 0.2815926515211539 0.8076923076923077
25 galtan overall All matched held-out ratings 2428 76 0.3879064541660041 0.18914006264912334 0.23816373704876254 -0.013710351768187365 0.4126853377265239 0.39325537210953215 0.4323911667110884 0.258665849890383 0.0914332784184514
26 galtan decade 1970s 287 23 0.2402564326635899 0.21665443860032066 0.25724972613838026 0.0744907055227787 0.3344947735191638 0.2824119799338199 0.3909497399852015 0.26474203940810126 0.0
27 galtan decade 2010s 693 57 0.4606647139522965 0.1935411541181932 0.24590414523794452 -0.050660741407812654 0.3838383838383838 0.3483644977339694 0.4205930386702865 0.24824147265028856 0.12265512265512266
28 galtan decade 2000s 676 55 0.3870265819854018 0.18629380051270433 0.23690167706252432 -0.0468645901306479 0.4275147928994083 0.39073352449867466 0.4651152496863708 0.258417938360392 0.11538461538461539
29 galtan decade 1990s 499 50 0.41212741006828396 0.17126387793509984 0.2212178226996366 -0.019602997434269034 0.48897795591182364 0.4453697785277533 0.5327545453149837 0.26606359520380773 0.11823647294589178
30 galtan decade 1980s 266 20 0.3476988123327735 0.18921982830486436 0.2308346130602219 0.08154082475673356 0.37969924812030076 0.3234807111340447 0.4393431084697526 0.2599239549642858 0.0
31 galtan decade 2020s 7 5 0.6324675234649824 0.17149569493926073 0.20386312323528286 0.030403782877832106 0.8571428571428571 0.48686549668097007 0.9743210440510253 0.49033504546428575 0.0
32 galtan region Asia-Pacific 142 5 -0.06990999699587813 0.2028535844559974 0.2421448847066388 -0.04764709776122232 0.43661971830985913 0.3577770958258817 0.5188013289856945 0.31393695714260567 0.007042253521126761
33 galtan region Europe 1938 59 0.4006536844413476 0.18821986618869885 0.2391185533726696 -0.007048457123688323 0.4107327141382869 0.3890267037664939 0.4327919244882515 0.24984948892156866 0.07791537667698659
34 galtan region North America 117 3 0.15158098770839884 0.2020264962636105 0.24484307272242728 0.04760720430768876 0.2905982905982906 0.21602760266916315 0.37848289702630594 0.23609932028311967 0.0
35 galtan region Latin America 110 4 0.510877747351673 0.15390455215418858 0.19797883572628092 -0.07917475272712499 0.6272727272727273 0.5340503169531756 0.7119054672241376 0.3662500658409092 0.6363636363636364
36 galtan region Other 121 5 0.3370413394210674 0.20735670781668103 0.24493355112293227 -0.08036162321796078 0.33884297520661155 0.2606254540535898 0.4269786779024248 0.25902643284276866 0.0
37 galtan text_distance_class direct text 2293 74 0.3817373914639877 0.1904499145514716 0.23993647489228756 -0.014524078018235874 0.404709986916703 0.38479506099767413 0.42494366891735347 0.25436052910535323 0.0
38 galtan text_distance_class nearby text (1--3 years) 135 40 0.4399561874640988 0.16689198552257195 0.20573340547684066 0.00011093927893245328 0.5481481481481482 0.46402181812792126 0.6296100616545071 0.3317925207057409 1.6444444444444444
39 galtan project V-Party 2273 70 0.3740693791970041 0.1911074338064693 0.24060334153635682 -0.018396438536216624 0.4047514298284206 0.3847495239974894 0.4250747518765987 0.2566477345279916 0.04399472063352398
40 galtan project GPS 26 26 0.5443693144157825 0.17671657291477977 0.21780593561007713 0.012007051328791764 0.46153846153846156 0.28755582695234905 0.6454236379556988 0.2777555471798077 0.8076923076923077
41 galtan project CHES 129 31 0.5869373239223193 0.15697863700916778 0.19496790336712624 0.06367587104738649 0.5426356589147286 0.456675986048851 0.6261294002156924 0.2903778195740311 0.7829457364341085
42 galtan item culsup_vparty 457 70 0.5366188718286774 0.16311644869435024 0.21359963974416982 0.011168223905568687 0.4638949671772429 0.418663126356892 0.5097287549316027 0.25691081806515326 0.0437636761487965
43 galtan item gender_vparty 445 70 0.32459509946925547 0.2164337137525867 0.2608755427368426 0.12306040878157333 0.34606741573033706 0.3033546905138159 0.39141513474302303 0.25556702282926974 0.0449438202247191
44 galtan item immig_vparty 457 70 0.5004863614572931 0.13311846254318913 0.16732357853899973 -0.011445561652418225 0.5776805251641138 0.5319316389137728 0.6221343134655263 0.25691081806515326 0.0437636761487965
45 galtan item lgbt_vparty 457 70 0.46354086621523605 0.15471230984590562 0.18981764808744647 0.04310285191432138 0.474835886214442 0.42945203114163966 0.5206392800594324 0.25691081806515326 0.0437636761487965
46 galtan item relig_vparty 457 70 0.4185372751933906 0.288821256864484 0.33467597534992094 -0.25415371263710096 0.15973741794310722 0.12900420700742332 0.19614352271142144 0.25691081806515326 0.0437636761487965
47 galtan item libcon_gps 26 26 0.5443693144157825 0.17671657291477977 0.21780593561007713 0.012007051328791764 0.46153846153846156 0.28755582695234905 0.6454236379556988 0.2777555471798077 0.8076923076923077
48 galtan item galtan_ches 129 31 0.5869373239223193 0.15697863700916778 0.19496790336712624 0.06367587104738649 0.5426356589147286 0.456675986048851 0.6261294002156924 0.2903778195740311 0.7829457364341085
@@ -0,0 +1,18 @@
"party_id","target_year","label","status","actual_year","party_name","country","economic_lr","economic_lower","economic_upper","galtan","galtan_lower","galtan_upper","source_support_class","era"
383,1972,"SPD 1972","exact",1972,"Social Democratic Party of Germany","DE",0.271573270375,0.162936225,0.38761415,0.370199619125,0.2680677,0.473299475,"both_direct_or_nearby","Historical & Cold War Era (1970–1999)"
383,2021,"SPD 2021","exact",2021,"Social Democratic Party of Germany","DE",0.2363366825,0.148430575,0.330595375,0.36824540875,0.27656695,0.461600275,"both_direct_or_nearby","Contemporary Era (2000–2022)"
1375,1983,"CDU 1983","exact",1983,"Christian Democratic Union","DE",0.704186118875,0.559885375,0.832835125,0.6167079735,0.5088347,0.716463575,"text_only_direct_or_nearby","Historical & Cold War Era (1970–1999)"
1375,2021,"CDU 2021","exact",2021,"Christian Democratic Union","DE",0.57714137175,0.45681705,0.69505505,0.536185121875,0.433751725,0.636189175,"text_only_direct_or_nearby","Contemporary Era (2000–2022)"
1516,1983,"Labour 1983","exact",1983,"Labour Party","GB",0.259080600525,0.1453436,0.3814767,0.34731692875,0.243851525,0.45443645,"both_direct_or_nearby","Historical & Cold War Era (1970–1999)"
1516,1997,"Labour 1997","exact",1997,"Labour Party","GB",0.540093913375,0.45312245,0.63004375,0.50597976425,0.44681665,0.566031675,"both_direct_or_nearby","Historical & Cold War Era (1970–1999)"
1567,1979,"Conservatives 1979","exact",1979,"Conservative Party","GB",0.863733304625,0.78370345,0.932120225,0.58421918175,0.47261405,0.691922375,"both_direct_or_nearby","Historical & Cold War Era (1970–1999)"
1567,2019,"Conservatives 2019","exact",2019,"Conservative Party","GB",0.734806347125,0.6674518,0.805338225,0.604817389125,0.5601045,0.6518521,"both_direct_or_nearby","Contemporary Era (2000–2022)"
487,1994,"Swedish SAP 1994","exact",1994,"Social Democratic Labour Party","SE",0.58329747225,0.45880465,0.706358175,0.323307833,0.22678185,0.42106565,"both_direct_or_nearby","Historical & Cold War Era (1970–1999)"
487,2022,"Swedish SAP 2022","exact",2022,"Social Democratic Labour Party","SE",0.271161611625,0.18001335,0.364595125,0.431600183875,0.333020475,0.530597425,"both_direct_or_nearby","Contemporary Era (2000–2022)"
409,2010,"Sweden Democrats 2010","exact",2010,"Sweden Democrats","SE",0.57402493025,0.47088145,0.681297275,0.721373298875,0.653960175,0.78768305,"both_direct_or_nearby","Contemporary Era (2000–2022)"
409,2022,"Sweden Democrats 2022","exact",2022,"Sweden Democrats","SE",0.609638368125,0.490017925,0.7238046,0.753425659,0.653108875,0.84672335,"both_direct_or_nearby","Contemporary Era (2000–2022)"
433,1988,"French FN 1988","exact",1988,"National Front","FR",0.78687783275,0.6924592,0.8758855,0.797982186375,0.7315981,0.86105335,"both_direct_or_nearby","Historical & Cold War Era (1970–1999)"
433,2022,"French FN 2022","exact",2022,"National Front","FR",0.524320955125,0.398589175,0.645052275,0.829604491,0.73907015,0.911430725,"both_direct_or_nearby","Contemporary Era (2000–2022)"
1545,2021,"The Left 2021","exact",2021,"The Left","DE",0.05463340125125,0.0205853625,0.104672875,0.2266517688,0.13204575,0.336095325,"both_direct_or_nearby","Contemporary Era (2000–2022)"
432,2020,"US Democrats 2020","exact",2020,"Democratic Party","US",0.299892256625,0.1963003,0.410606025,0.3345769065,0.263694925,0.405684575,"both_direct_or_nearby","Contemporary Era (2000–2022)"
809,2020,"US Republicans 2020","exact",2020,"Republican Party","US",0.8507495345,0.77742365,0.917357075,0.721311676375,0.65728435,0.78442835,"both_direct_or_nearby","Contemporary Era (2000–2022)"
1 party_id target_year label status actual_year party_name country economic_lr economic_lower economic_upper galtan galtan_lower galtan_upper source_support_class era
2 383 1972 SPD 1972 exact 1972 Social Democratic Party of Germany DE 0.271573270375 0.162936225 0.38761415 0.370199619125 0.2680677 0.473299475 both_direct_or_nearby Historical & Cold War Era (1970–1999)
3 383 2021 SPD 2021 exact 2021 Social Democratic Party of Germany DE 0.2363366825 0.148430575 0.330595375 0.36824540875 0.27656695 0.461600275 both_direct_or_nearby Contemporary Era (2000–2022)
4 1375 1983 CDU 1983 exact 1983 Christian Democratic Union DE 0.704186118875 0.559885375 0.832835125 0.6167079735 0.5088347 0.716463575 text_only_direct_or_nearby Historical & Cold War Era (1970–1999)
5 1375 2021 CDU 2021 exact 2021 Christian Democratic Union DE 0.57714137175 0.45681705 0.69505505 0.536185121875 0.433751725 0.636189175 text_only_direct_or_nearby Contemporary Era (2000–2022)
6 1516 1983 Labour 1983 exact 1983 Labour Party GB 0.259080600525 0.1453436 0.3814767 0.34731692875 0.243851525 0.45443645 both_direct_or_nearby Historical & Cold War Era (1970–1999)
7 1516 1997 Labour 1997 exact 1997 Labour Party GB 0.540093913375 0.45312245 0.63004375 0.50597976425 0.44681665 0.566031675 both_direct_or_nearby Historical & Cold War Era (1970–1999)
8 1567 1979 Conservatives 1979 exact 1979 Conservative Party GB 0.863733304625 0.78370345 0.932120225 0.58421918175 0.47261405 0.691922375 both_direct_or_nearby Historical & Cold War Era (1970–1999)
9 1567 2019 Conservatives 2019 exact 2019 Conservative Party GB 0.734806347125 0.6674518 0.805338225 0.604817389125 0.5601045 0.6518521 both_direct_or_nearby Contemporary Era (2000–2022)
10 487 1994 Swedish SAP 1994 exact 1994 Social Democratic Labour Party SE 0.58329747225 0.45880465 0.706358175 0.323307833 0.22678185 0.42106565 both_direct_or_nearby Historical & Cold War Era (1970–1999)
11 487 2022 Swedish SAP 2022 exact 2022 Social Democratic Labour Party SE 0.271161611625 0.18001335 0.364595125 0.431600183875 0.333020475 0.530597425 both_direct_or_nearby Contemporary Era (2000–2022)
12 409 2010 Sweden Democrats 2010 exact 2010 Sweden Democrats SE 0.57402493025 0.47088145 0.681297275 0.721373298875 0.653960175 0.78768305 both_direct_or_nearby Contemporary Era (2000–2022)
13 409 2022 Sweden Democrats 2022 exact 2022 Sweden Democrats SE 0.609638368125 0.490017925 0.7238046 0.753425659 0.653108875 0.84672335 both_direct_or_nearby Contemporary Era (2000–2022)
14 433 1988 French FN 1988 exact 1988 National Front FR 0.78687783275 0.6924592 0.8758855 0.797982186375 0.7315981 0.86105335 both_direct_or_nearby Historical & Cold War Era (1970–1999)
15 433 2022 French FN 2022 exact 2022 National Front FR 0.524320955125 0.398589175 0.645052275 0.829604491 0.73907015 0.911430725 both_direct_or_nearby Contemporary Era (2000–2022)
16 1545 2021 The Left 2021 exact 2021 The Left DE 0.05463340125125 0.0205853625 0.104672875 0.2266517688 0.13204575 0.336095325 both_direct_or_nearby Contemporary Era (2000–2022)
17 432 2020 US Democrats 2020 exact 2020 Democratic Party US 0.299892256625 0.1963003 0.410606025 0.3345769065 0.263694925 0.405684575 both_direct_or_nearby Contemporary Era (2000–2022)
18 809 2020 US Republicans 2020 exact 2020 Republican Party US 0.8507495345 0.77742365 0.917357075 0.721311676375 0.65728435 0.78442835 both_direct_or_nearby Contemporary Era (2000–2022)
@@ -0,0 +1,147 @@
"party_id","party_label","country","year","source_support_class","dimension","estimate","lower","upper"
1567,"United Kingdom: Conservatives","GB",1945,"text_only_direct_or_nearby","Economic",0.7897200205,0.68672145,0.882517725
1567,"United Kingdom: Conservatives","GB",1950,"text_only_direct_or_nearby","Economic",0.696324671375,0.58645,0.7999215
1567,"United Kingdom: Conservatives","GB",1951,"text_only_direct_or_nearby","Economic",0.648801438125,0.509441425,0.7756712
1567,"United Kingdom: Conservatives","GB",1955,"text_only_direct_or_nearby","Economic",0.559596760875,0.4111588,0.696031575
1567,"United Kingdom: Conservatives","GB",1959,"text_only_direct_or_nearby","Economic",0.56006642225,0.388645375,0.718084975
1567,"United Kingdom: Conservatives","GB",1964,"text_only_direct_or_nearby","Economic",0.713843190625,0.5883803,0.829548525
1567,"United Kingdom: Conservatives","GB",1966,"text_only_direct_or_nearby","Economic",0.79834168225,0.6939884,0.890500425
1567,"United Kingdom: Conservatives","GB",1970,"both_direct_or_nearby","Economic",0.67897010575,0.55999,0.79320825
1567,"United Kingdom: Conservatives","GB",1974,"both_direct_or_nearby","Economic",0.68530368975,0.6022573,0.771763125
1567,"United Kingdom: Conservatives","GB",1979,"both_direct_or_nearby","Economic",0.863733304625,0.78370345,0.932120225
1567,"United Kingdom: Conservatives","GB",1983,"both_direct_or_nearby","Economic",0.897143095,0.823332125,0.953360675
1567,"United Kingdom: Conservatives","GB",1987,"both_direct_or_nearby","Economic",0.864987285375,0.782492075,0.9328753
1567,"United Kingdom: Conservatives","GB",1992,"both_direct_or_nearby","Economic",0.80683245025,0.731234925,0.88153825
1567,"United Kingdom: Conservatives","GB",1997,"both_direct_or_nearby","Economic",0.815790791,0.741322225,0.886000175
1567,"United Kingdom: Conservatives","GB",2001,"both_direct_or_nearby","Economic",0.735943766375,0.64358755,0.825612475
1567,"United Kingdom: Conservatives","GB",2005,"both_direct_or_nearby","Economic",0.7468415025,0.657667875,0.8347423
1567,"United Kingdom: Conservatives","GB",2010,"both_direct_or_nearby","Economic",0.807639387875,0.737347625,0.877409025
1567,"United Kingdom: Conservatives","GB",2015,"both_direct_or_nearby","Economic",0.650294524875,0.582851025,0.72313155
1567,"United Kingdom: Conservatives","GB",2017,"both_direct_or_nearby","Economic",0.680683625125,0.607886575,0.758350075
1567,"United Kingdom: Conservatives","GB",2019,"both_direct_or_nearby","Economic",0.734806347125,0.6674518,0.805338225
1567,"United Kingdom: Conservatives","GB",2024,"both_direct_or_nearby","Economic",0.696441988625,0.611536275,0.7845477
383,"Germany: SPD","DE",1949,"text_only_direct_or_nearby","Economic",0.2594422284125,0.124083075,0.416025325
383,"Germany: SPD","DE",1953,"text_only_direct_or_nearby","Economic",0.3044620151625,0.172417475,0.44553015
383,"Germany: SPD","DE",1957,"text_only_direct_or_nearby","Economic",0.364714905375,0.217271,0.518439625
383,"Germany: SPD","DE",1961,"text_only_direct_or_nearby","Economic",0.45071135275,0.3107251,0.58867
383,"Germany: SPD","DE",1965,"text_only_direct_or_nearby","Economic",0.40004889475,0.253314775,0.5547562
383,"Germany: SPD","DE",1969,"text_only_direct_or_nearby","Economic",0.36430581075,0.234999825,0.496653
383,"Germany: SPD","DE",1972,"both_direct_or_nearby","Economic",0.271573270375,0.162936225,0.38761415
383,"Germany: SPD","DE",1976,"both_direct_or_nearby","Economic",0.22760841445,0.1408578,0.317522325
383,"Germany: SPD","DE",1980,"both_direct_or_nearby","Economic",0.2854944402625,0.176353725,0.402044875
383,"Germany: SPD","DE",1983,"both_direct_or_nearby","Economic",0.3201057485,0.198932275,0.449359525
383,"Germany: SPD","DE",1987,"both_direct_or_nearby","Economic",0.3082503073875,0.1890146,0.43586885
383,"Germany: SPD","DE",1990,"both_direct_or_nearby","Economic",0.2997481731625,0.1826493,0.42186305
383,"Germany: SPD","DE",1994,"both_direct_or_nearby","Economic",0.344258320125,0.247450525,0.440361825
383,"Germany: SPD","DE",1998,"both_direct_or_nearby","Economic",0.42037938775,0.335025575,0.504438225
383,"Germany: SPD","DE",2002,"both_direct_or_nearby","Economic",0.354348980875,0.285402475,0.419218875
383,"Germany: SPD","DE",2005,"both_direct_or_nearby","Economic",0.408510935875,0.326714325,0.4912897
383,"Germany: SPD","DE",2009,"both_direct_or_nearby","Economic",0.2258038923125,0.154613,0.2986832
383,"Germany: SPD","DE",2013,"both_direct_or_nearby","Economic",0.2547526295,0.177729475,0.3305311
383,"Germany: SPD","DE",2017,"both_direct_or_nearby","Economic",0.2187940564125,0.150431,0.288139
383,"Germany: SPD","DE",2021,"both_direct_or_nearby","Economic",0.2363366825,0.148430575,0.330595375
383,"Germany: SPD","DE",2025,"both_direct_or_nearby","Economic",0.24988987125,0.155485225,0.353246025
379,"Denmark: Social Democrats","DK",1945,"text_only_direct_or_nearby","Economic",0.39289630125,0.306313775,0.4744366
379,"Denmark: Social Democrats","DK",1947,"text_only_direct_or_nearby","Economic",0.4133295765,0.326647125,0.497578325
379,"Denmark: Social Democrats","DK",1950,"text_only_direct_or_nearby","Economic",0.402314539625,0.313989,0.48990055
379,"Denmark: Social Democrats","DK",1953,"text_only_direct_or_nearby","Economic",0.404059824625,0.31188405,0.492347325
379,"Denmark: Social Democrats","DK",1957,"text_only_direct_or_nearby","Economic",0.386770333875,0.2915608,0.4740622
379,"Denmark: Social Democrats","DK",1960,"text_only_direct_or_nearby","Economic",0.39128767025,0.30583805,0.46915175
379,"Denmark: Social Democrats","DK",1964,"text_only_direct_or_nearby","Economic",0.41323511225,0.3319368,0.490628525
379,"Denmark: Social Democrats","DK",1966,"text_only_direct_or_nearby","Economic",0.424408032125,0.348613025,0.502216325
379,"Denmark: Social Democrats","DK",1968,"text_only_direct_or_nearby","Economic",0.426020581125,0.346608825,0.502008425
379,"Denmark: Social Democrats","DK",1971,"both_direct_or_nearby","Economic",0.3838693815,0.31048555,0.4534922
379,"Denmark: Social Democrats","DK",1973,"both_direct_or_nearby","Economic",0.36916859375,0.292048475,0.4383149
379,"Denmark: Social Democrats","DK",1975,"both_direct_or_nearby","Economic",0.2551262250375,0.1655264,0.348229075
379,"Denmark: Social Democrats","DK",1977,"both_direct_or_nearby","Economic",0.2030203874375,0.11497425,0.301540125
379,"Denmark: Social Democrats","DK",1979,"both_direct_or_nearby","Economic",0.1862332029375,0.10144465,0.2859892
379,"Denmark: Social Democrats","DK",1981,"both_direct_or_nearby","Economic",0.1919036092625,0.1020977,0.2975764
379,"Denmark: Social Democrats","DK",1984,"both_direct_or_nearby","Economic",0.1734391782625,0.08982039,0.2725962
379,"Denmark: Social Democrats","DK",1987,"both_direct_or_nearby","Economic",0.1872400464625,0.10188485,0.281978775
379,"Denmark: Social Democrats","DK",1988,"both_direct_or_nearby","Economic",0.1844144026,0.10014465,0.278666375
379,"Denmark: Social Democrats","DK",1990,"both_direct_or_nearby","Economic",0.1593657011125,0.0820198525,0.250674325
379,"Denmark: Social Democrats","DK",1994,"both_direct_or_nearby","Economic",0.18027195025,0.0911998025,0.286319075
379,"Denmark: Social Democrats","DK",1998,"both_direct_or_nearby","Economic",0.2332968959625,0.14133775,0.3324363
379,"Denmark: Social Democrats","DK",2001,"both_direct_or_nearby","Economic",0.25665459875,0.167550175,0.3497182
379,"Denmark: Social Democrats","DK",2005,"both_direct_or_nearby","Economic",0.243111738,0.15668705,0.33332095
379,"Denmark: Social Democrats","DK",2007,"both_direct_or_nearby","Economic",0.2504832553,0.1616851,0.347202325
379,"Denmark: Social Democrats","DK",2011,"both_direct_or_nearby","Economic",0.277551306,0.187687975,0.3698644
379,"Denmark: Social Democrats","DK",2015,"both_direct_or_nearby","Economic",0.2492670365,0.160125675,0.34358355
379,"Denmark: Social Democrats","DK",2019,"both_direct_or_nearby","Economic",0.254542113875,0.180740975,0.32727345
409,"Sweden: Sweden Democrats","SE",2010,"both_direct_or_nearby","Economic",0.57402493025,0.47088145,0.681297275
409,"Sweden: Sweden Democrats","SE",2014,"both_direct_or_nearby","Economic",0.514751535125,0.4326713,0.602047575
409,"Sweden: Sweden Democrats","SE",2018,"both_direct_or_nearby","Economic",0.499110805875,0.397517425,0.59893585
409,"Sweden: Sweden Democrats","SE",2022,"both_direct_or_nearby","Economic",0.609638368125,0.490017925,0.7238046
1567,"United Kingdom: Conservatives","GB",1945,"text_only_direct_or_nearby","Cultural",0.42283955775,0.30617415,0.539308425
1567,"United Kingdom: Conservatives","GB",1950,"text_only_direct_or_nearby","Cultural",0.597652829625,0.493804425,0.695553875
1567,"United Kingdom: Conservatives","GB",1951,"text_only_direct_or_nearby","Cultural",0.5963407815,0.4953524,0.694196075
1567,"United Kingdom: Conservatives","GB",1955,"text_only_direct_or_nearby","Cultural",0.4481433145,0.349619975,0.546246075
1567,"United Kingdom: Conservatives","GB",1959,"text_only_direct_or_nearby","Cultural",0.379823589625,0.24070795,0.5271287
1567,"United Kingdom: Conservatives","GB",1964,"text_only_direct_or_nearby","Cultural",0.4359372675,0.309309225,0.5597016
1567,"United Kingdom: Conservatives","GB",1966,"text_only_direct_or_nearby","Cultural",0.433826746,0.305561125,0.564284875
1567,"United Kingdom: Conservatives","GB",1970,"both_direct_or_nearby","Cultural",0.586682156125,0.487144225,0.681696075
1567,"United Kingdom: Conservatives","GB",1974,"both_direct_or_nearby","Cultural",0.582726903625,0.512885,0.65400205
1567,"United Kingdom: Conservatives","GB",1979,"both_direct_or_nearby","Cultural",0.58421918175,0.47261405,0.691922375
1567,"United Kingdom: Conservatives","GB",1983,"both_direct_or_nearby","Cultural",0.54862467925,0.4436958,0.651204175
1567,"United Kingdom: Conservatives","GB",1987,"both_direct_or_nearby","Cultural",0.5044723465,0.398577575,0.60606875
1567,"United Kingdom: Conservatives","GB",1992,"both_direct_or_nearby","Cultural",0.608807442,0.53829935,0.67952055
1567,"United Kingdom: Conservatives","GB",1997,"both_direct_or_nearby","Cultural",0.653249609625,0.5903779,0.719252725
1567,"United Kingdom: Conservatives","GB",2001,"both_direct_or_nearby","Cultural",0.62022832225,0.5519112,0.68966615
1567,"United Kingdom: Conservatives","GB",2005,"both_direct_or_nearby","Cultural",0.63051460775,0.562743875,0.6984813
1567,"United Kingdom: Conservatives","GB",2010,"both_direct_or_nearby","Cultural",0.515056946125,0.464064225,0.5684148
1567,"United Kingdom: Conservatives","GB",2015,"both_direct_or_nearby","Cultural",0.53108352025,0.47741145,0.585770875
1567,"United Kingdom: Conservatives","GB",2017,"both_direct_or_nearby","Cultural",0.550617655875,0.503622825,0.597269
1567,"United Kingdom: Conservatives","GB",2019,"both_direct_or_nearby","Cultural",0.604817389125,0.5601045,0.6518521
1567,"United Kingdom: Conservatives","GB",2024,"both_direct_or_nearby","Cultural",0.64147235475,0.568830525,0.717038025
383,"Germany: SPD","DE",1949,"text_only_direct_or_nearby","Cultural",0.4562653665,0.29105445,0.6148909
383,"Germany: SPD","DE",1953,"text_only_direct_or_nearby","Cultural",0.41201424225,0.239960775,0.59720855
383,"Germany: SPD","DE",1957,"text_only_direct_or_nearby","Cultural",0.361933815225,0.2020638,0.530546125
383,"Germany: SPD","DE",1961,"text_only_direct_or_nearby","Cultural",0.3403875836625,0.189634975,0.502483725
383,"Germany: SPD","DE",1965,"text_only_direct_or_nearby","Cultural",0.349500525125,0.197610625,0.519782425
383,"Germany: SPD","DE",1969,"text_only_direct_or_nearby","Cultural",0.367453271375,0.2312429,0.51644905
383,"Germany: SPD","DE",1972,"both_direct_or_nearby","Cultural",0.370199619125,0.2680677,0.473299475
383,"Germany: SPD","DE",1976,"both_direct_or_nearby","Cultural",0.28787436475,0.21332925,0.364130925
383,"Germany: SPD","DE",1980,"both_direct_or_nearby","Cultural",0.346500680875,0.245533525,0.448428875
383,"Germany: SPD","DE",1983,"both_direct_or_nearby","Cultural",0.3643932085,0.2585273,0.474535775
383,"Germany: SPD","DE",1987,"both_direct_or_nearby","Cultural",0.373242374375,0.266124675,0.479610175
383,"Germany: SPD","DE",1990,"both_direct_or_nearby","Cultural",0.354967389625,0.25200245,0.460687325
383,"Germany: SPD","DE",1994,"both_direct_or_nearby","Cultural",0.410000861625,0.3313879,0.488224525
383,"Germany: SPD","DE",1998,"both_direct_or_nearby","Cultural",0.458140040875,0.389836225,0.525399375
383,"Germany: SPD","DE",2002,"both_direct_or_nearby","Cultural",0.442940016,0.3876151,0.4964075
383,"Germany: SPD","DE",2005,"both_direct_or_nearby","Cultural",0.456470195875,0.39176415,0.5189629
383,"Germany: SPD","DE",2009,"both_direct_or_nearby","Cultural",0.389663805625,0.32557465,0.451253625
383,"Germany: SPD","DE",2013,"both_direct_or_nearby","Cultural",0.2109102855,0.15503785,0.268407625
383,"Germany: SPD","DE",2017,"both_direct_or_nearby","Cultural",0.40809383225,0.35346185,0.45866175
383,"Germany: SPD","DE",2021,"both_direct_or_nearby","Cultural",0.36824540875,0.27656695,0.461600275
383,"Germany: SPD","DE",2025,"both_direct_or_nearby","Cultural",0.3699090585,0.27155195,0.46999925
379,"Denmark: Social Democrats","DK",1945,"text_only_direct_or_nearby","Cultural",0.367253073125,0.222935775,0.513292475
379,"Denmark: Social Democrats","DK",1947,"text_only_direct_or_nearby","Cultural",0.38745918675,0.234832075,0.545806525
379,"Denmark: Social Democrats","DK",1950,"text_only_direct_or_nearby","Cultural",0.3968814625,0.2531828,0.544707225
379,"Denmark: Social Democrats","DK",1953,"text_only_direct_or_nearby","Cultural",0.42331649125,0.283525625,0.5604742
379,"Denmark: Social Democrats","DK",1957,"text_only_direct_or_nearby","Cultural",0.42351804925,0.268476675,0.589941825
379,"Denmark: Social Democrats","DK",1960,"text_only_direct_or_nearby","Cultural",0.384011194875,0.24432495,0.52920335
379,"Denmark: Social Democrats","DK",1964,"text_only_direct_or_nearby","Cultural",0.362245514375,0.224588175,0.502448675
379,"Denmark: Social Democrats","DK",1966,"text_only_direct_or_nearby","Cultural",0.3679157005,0.230511325,0.5031338
379,"Denmark: Social Democrats","DK",1968,"text_only_direct_or_nearby","Cultural",0.382175081625,0.25659165,0.507958075
379,"Denmark: Social Democrats","DK",1971,"both_direct_or_nearby","Cultural",0.470225764375,0.36279785,0.577130125
379,"Denmark: Social Democrats","DK",1973,"both_direct_or_nearby","Cultural",0.501466689875,0.3986878,0.60298055
379,"Denmark: Social Democrats","DK",1975,"both_direct_or_nearby","Cultural",0.51072920025,0.41364105,0.608167125
379,"Denmark: Social Democrats","DK",1977,"both_direct_or_nearby","Cultural",0.48888551625,0.3865325,0.588461525
379,"Denmark: Social Democrats","DK",1979,"both_direct_or_nearby","Cultural",0.469331547625,0.36461055,0.5720459
379,"Denmark: Social Democrats","DK",1981,"both_direct_or_nearby","Cultural",0.440505821125,0.334911475,0.544872125
379,"Denmark: Social Democrats","DK",1984,"both_direct_or_nearby","Cultural",0.427927746125,0.3199225,0.535149325
379,"Denmark: Social Democrats","DK",1987,"both_direct_or_nearby","Cultural",0.42510295375,0.332117075,0.518948075
379,"Denmark: Social Democrats","DK",1988,"both_direct_or_nearby","Cultural",0.4189467795,0.3290549,0.51075845
379,"Denmark: Social Democrats","DK",1990,"both_direct_or_nearby","Cultural",0.394384283125,0.289929725,0.498693825
379,"Denmark: Social Democrats","DK",1994,"both_direct_or_nearby","Cultural",0.365051991,0.259654975,0.469250275
379,"Denmark: Social Democrats","DK",1998,"both_direct_or_nearby","Cultural",0.4057949125,0.318039475,0.493913625
379,"Denmark: Social Democrats","DK",2001,"both_direct_or_nearby","Cultural",0.377267655375,0.2986888,0.456555875
379,"Denmark: Social Democrats","DK",2005,"both_direct_or_nearby","Cultural",0.3840174395,0.296652025,0.472492425
379,"Denmark: Social Democrats","DK",2007,"both_direct_or_nearby","Cultural",0.377073825875,0.288222975,0.46642125
379,"Denmark: Social Democrats","DK",2011,"both_direct_or_nearby","Cultural",0.44217709925,0.351102525,0.5332076
379,"Denmark: Social Democrats","DK",2015,"both_direct_or_nearby","Cultural",0.5274942445,0.44580275,0.608968575
379,"Denmark: Social Democrats","DK",2019,"both_direct_or_nearby","Cultural",0.440971318625,0.38293125,0.4998911
409,"Sweden: Sweden Democrats","SE",2010,"both_direct_or_nearby","Cultural",0.721373298875,0.653960175,0.78768305
409,"Sweden: Sweden Democrats","SE",2014,"both_direct_or_nearby","Cultural",0.750375772125,0.690054075,0.81067825
409,"Sweden: Sweden Democrats","SE",2018,"both_direct_or_nearby","Cultural",0.671777804625,0.5936524,0.747156725
409,"Sweden: Sweden Democrats","SE",2022,"both_direct_or_nearby","Cultural",0.753425659,0.653108875,0.84672335
1 party_id party_label country year source_support_class dimension estimate lower upper
2 1567 United Kingdom: Conservatives GB 1945 text_only_direct_or_nearby Economic 0.7897200205 0.68672145 0.882517725
3 1567 United Kingdom: Conservatives GB 1950 text_only_direct_or_nearby Economic 0.696324671375 0.58645 0.7999215
4 1567 United Kingdom: Conservatives GB 1951 text_only_direct_or_nearby Economic 0.648801438125 0.509441425 0.7756712
5 1567 United Kingdom: Conservatives GB 1955 text_only_direct_or_nearby Economic 0.559596760875 0.4111588 0.696031575
6 1567 United Kingdom: Conservatives GB 1959 text_only_direct_or_nearby Economic 0.56006642225 0.388645375 0.718084975
7 1567 United Kingdom: Conservatives GB 1964 text_only_direct_or_nearby Economic 0.713843190625 0.5883803 0.829548525
8 1567 United Kingdom: Conservatives GB 1966 text_only_direct_or_nearby Economic 0.79834168225 0.6939884 0.890500425
9 1567 United Kingdom: Conservatives GB 1970 both_direct_or_nearby Economic 0.67897010575 0.55999 0.79320825
10 1567 United Kingdom: Conservatives GB 1974 both_direct_or_nearby Economic 0.68530368975 0.6022573 0.771763125
11 1567 United Kingdom: Conservatives GB 1979 both_direct_or_nearby Economic 0.863733304625 0.78370345 0.932120225
12 1567 United Kingdom: Conservatives GB 1983 both_direct_or_nearby Economic 0.897143095 0.823332125 0.953360675
13 1567 United Kingdom: Conservatives GB 1987 both_direct_or_nearby Economic 0.864987285375 0.782492075 0.9328753
14 1567 United Kingdom: Conservatives GB 1992 both_direct_or_nearby Economic 0.80683245025 0.731234925 0.88153825
15 1567 United Kingdom: Conservatives GB 1997 both_direct_or_nearby Economic 0.815790791 0.741322225 0.886000175
16 1567 United Kingdom: Conservatives GB 2001 both_direct_or_nearby Economic 0.735943766375 0.64358755 0.825612475
17 1567 United Kingdom: Conservatives GB 2005 both_direct_or_nearby Economic 0.7468415025 0.657667875 0.8347423
18 1567 United Kingdom: Conservatives GB 2010 both_direct_or_nearby Economic 0.807639387875 0.737347625 0.877409025
19 1567 United Kingdom: Conservatives GB 2015 both_direct_or_nearby Economic 0.650294524875 0.582851025 0.72313155
20 1567 United Kingdom: Conservatives GB 2017 both_direct_or_nearby Economic 0.680683625125 0.607886575 0.758350075
21 1567 United Kingdom: Conservatives GB 2019 both_direct_or_nearby Economic 0.734806347125 0.6674518 0.805338225
22 1567 United Kingdom: Conservatives GB 2024 both_direct_or_nearby Economic 0.696441988625 0.611536275 0.7845477
23 383 Germany: SPD DE 1949 text_only_direct_or_nearby Economic 0.2594422284125 0.124083075 0.416025325
24 383 Germany: SPD DE 1953 text_only_direct_or_nearby Economic 0.3044620151625 0.172417475 0.44553015
25 383 Germany: SPD DE 1957 text_only_direct_or_nearby Economic 0.364714905375 0.217271 0.518439625
26 383 Germany: SPD DE 1961 text_only_direct_or_nearby Economic 0.45071135275 0.3107251 0.58867
27 383 Germany: SPD DE 1965 text_only_direct_or_nearby Economic 0.40004889475 0.253314775 0.5547562
28 383 Germany: SPD DE 1969 text_only_direct_or_nearby Economic 0.36430581075 0.234999825 0.496653
29 383 Germany: SPD DE 1972 both_direct_or_nearby Economic 0.271573270375 0.162936225 0.38761415
30 383 Germany: SPD DE 1976 both_direct_or_nearby Economic 0.22760841445 0.1408578 0.317522325
31 383 Germany: SPD DE 1980 both_direct_or_nearby Economic 0.2854944402625 0.176353725 0.402044875
32 383 Germany: SPD DE 1983 both_direct_or_nearby Economic 0.3201057485 0.198932275 0.449359525
33 383 Germany: SPD DE 1987 both_direct_or_nearby Economic 0.3082503073875 0.1890146 0.43586885
34 383 Germany: SPD DE 1990 both_direct_or_nearby Economic 0.2997481731625 0.1826493 0.42186305
35 383 Germany: SPD DE 1994 both_direct_or_nearby Economic 0.344258320125 0.247450525 0.440361825
36 383 Germany: SPD DE 1998 both_direct_or_nearby Economic 0.42037938775 0.335025575 0.504438225
37 383 Germany: SPD DE 2002 both_direct_or_nearby Economic 0.354348980875 0.285402475 0.419218875
38 383 Germany: SPD DE 2005 both_direct_or_nearby Economic 0.408510935875 0.326714325 0.4912897
39 383 Germany: SPD DE 2009 both_direct_or_nearby Economic 0.2258038923125 0.154613 0.2986832
40 383 Germany: SPD DE 2013 both_direct_or_nearby Economic 0.2547526295 0.177729475 0.3305311
41 383 Germany: SPD DE 2017 both_direct_or_nearby Economic 0.2187940564125 0.150431 0.288139
42 383 Germany: SPD DE 2021 both_direct_or_nearby Economic 0.2363366825 0.148430575 0.330595375
43 383 Germany: SPD DE 2025 both_direct_or_nearby Economic 0.24988987125 0.155485225 0.353246025
44 379 Denmark: Social Democrats DK 1945 text_only_direct_or_nearby Economic 0.39289630125 0.306313775 0.4744366
45 379 Denmark: Social Democrats DK 1947 text_only_direct_or_nearby Economic 0.4133295765 0.326647125 0.497578325
46 379 Denmark: Social Democrats DK 1950 text_only_direct_or_nearby Economic 0.402314539625 0.313989 0.48990055
47 379 Denmark: Social Democrats DK 1953 text_only_direct_or_nearby Economic 0.404059824625 0.31188405 0.492347325
48 379 Denmark: Social Democrats DK 1957 text_only_direct_or_nearby Economic 0.386770333875 0.2915608 0.4740622
49 379 Denmark: Social Democrats DK 1960 text_only_direct_or_nearby Economic 0.39128767025 0.30583805 0.46915175
50 379 Denmark: Social Democrats DK 1964 text_only_direct_or_nearby Economic 0.41323511225 0.3319368 0.490628525
51 379 Denmark: Social Democrats DK 1966 text_only_direct_or_nearby Economic 0.424408032125 0.348613025 0.502216325
52 379 Denmark: Social Democrats DK 1968 text_only_direct_or_nearby Economic 0.426020581125 0.346608825 0.502008425
53 379 Denmark: Social Democrats DK 1971 both_direct_or_nearby Economic 0.3838693815 0.31048555 0.4534922
54 379 Denmark: Social Democrats DK 1973 both_direct_or_nearby Economic 0.36916859375 0.292048475 0.4383149
55 379 Denmark: Social Democrats DK 1975 both_direct_or_nearby Economic 0.2551262250375 0.1655264 0.348229075
56 379 Denmark: Social Democrats DK 1977 both_direct_or_nearby Economic 0.2030203874375 0.11497425 0.301540125
57 379 Denmark: Social Democrats DK 1979 both_direct_or_nearby Economic 0.1862332029375 0.10144465 0.2859892
58 379 Denmark: Social Democrats DK 1981 both_direct_or_nearby Economic 0.1919036092625 0.1020977 0.2975764
59 379 Denmark: Social Democrats DK 1984 both_direct_or_nearby Economic 0.1734391782625 0.08982039 0.2725962
60 379 Denmark: Social Democrats DK 1987 both_direct_or_nearby Economic 0.1872400464625 0.10188485 0.281978775
61 379 Denmark: Social Democrats DK 1988 both_direct_or_nearby Economic 0.1844144026 0.10014465 0.278666375
62 379 Denmark: Social Democrats DK 1990 both_direct_or_nearby Economic 0.1593657011125 0.0820198525 0.250674325
63 379 Denmark: Social Democrats DK 1994 both_direct_or_nearby Economic 0.18027195025 0.0911998025 0.286319075
64 379 Denmark: Social Democrats DK 1998 both_direct_or_nearby Economic 0.2332968959625 0.14133775 0.3324363
65 379 Denmark: Social Democrats DK 2001 both_direct_or_nearby Economic 0.25665459875 0.167550175 0.3497182
66 379 Denmark: Social Democrats DK 2005 both_direct_or_nearby Economic 0.243111738 0.15668705 0.33332095
67 379 Denmark: Social Democrats DK 2007 both_direct_or_nearby Economic 0.2504832553 0.1616851 0.347202325
68 379 Denmark: Social Democrats DK 2011 both_direct_or_nearby Economic 0.277551306 0.187687975 0.3698644
69 379 Denmark: Social Democrats DK 2015 both_direct_or_nearby Economic 0.2492670365 0.160125675 0.34358355
70 379 Denmark: Social Democrats DK 2019 both_direct_or_nearby Economic 0.254542113875 0.180740975 0.32727345
71 409 Sweden: Sweden Democrats SE 2010 both_direct_or_nearby Economic 0.57402493025 0.47088145 0.681297275
72 409 Sweden: Sweden Democrats SE 2014 both_direct_or_nearby Economic 0.514751535125 0.4326713 0.602047575
73 409 Sweden: Sweden Democrats SE 2018 both_direct_or_nearby Economic 0.499110805875 0.397517425 0.59893585
74 409 Sweden: Sweden Democrats SE 2022 both_direct_or_nearby Economic 0.609638368125 0.490017925 0.7238046
75 1567 United Kingdom: Conservatives GB 1945 text_only_direct_or_nearby Cultural 0.42283955775 0.30617415 0.539308425
76 1567 United Kingdom: Conservatives GB 1950 text_only_direct_or_nearby Cultural 0.597652829625 0.493804425 0.695553875
77 1567 United Kingdom: Conservatives GB 1951 text_only_direct_or_nearby Cultural 0.5963407815 0.4953524 0.694196075
78 1567 United Kingdom: Conservatives GB 1955 text_only_direct_or_nearby Cultural 0.4481433145 0.349619975 0.546246075
79 1567 United Kingdom: Conservatives GB 1959 text_only_direct_or_nearby Cultural 0.379823589625 0.24070795 0.5271287
80 1567 United Kingdom: Conservatives GB 1964 text_only_direct_or_nearby Cultural 0.4359372675 0.309309225 0.5597016
81 1567 United Kingdom: Conservatives GB 1966 text_only_direct_or_nearby Cultural 0.433826746 0.305561125 0.564284875
82 1567 United Kingdom: Conservatives GB 1970 both_direct_or_nearby Cultural 0.586682156125 0.487144225 0.681696075
83 1567 United Kingdom: Conservatives GB 1974 both_direct_or_nearby Cultural 0.582726903625 0.512885 0.65400205
84 1567 United Kingdom: Conservatives GB 1979 both_direct_or_nearby Cultural 0.58421918175 0.47261405 0.691922375
85 1567 United Kingdom: Conservatives GB 1983 both_direct_or_nearby Cultural 0.54862467925 0.4436958 0.651204175
86 1567 United Kingdom: Conservatives GB 1987 both_direct_or_nearby Cultural 0.5044723465 0.398577575 0.60606875
87 1567 United Kingdom: Conservatives GB 1992 both_direct_or_nearby Cultural 0.608807442 0.53829935 0.67952055
88 1567 United Kingdom: Conservatives GB 1997 both_direct_or_nearby Cultural 0.653249609625 0.5903779 0.719252725
89 1567 United Kingdom: Conservatives GB 2001 both_direct_or_nearby Cultural 0.62022832225 0.5519112 0.68966615
90 1567 United Kingdom: Conservatives GB 2005 both_direct_or_nearby Cultural 0.63051460775 0.562743875 0.6984813
91 1567 United Kingdom: Conservatives GB 2010 both_direct_or_nearby Cultural 0.515056946125 0.464064225 0.5684148
92 1567 United Kingdom: Conservatives GB 2015 both_direct_or_nearby Cultural 0.53108352025 0.47741145 0.585770875
93 1567 United Kingdom: Conservatives GB 2017 both_direct_or_nearby Cultural 0.550617655875 0.503622825 0.597269
94 1567 United Kingdom: Conservatives GB 2019 both_direct_or_nearby Cultural 0.604817389125 0.5601045 0.6518521
95 1567 United Kingdom: Conservatives GB 2024 both_direct_or_nearby Cultural 0.64147235475 0.568830525 0.717038025
96 383 Germany: SPD DE 1949 text_only_direct_or_nearby Cultural 0.4562653665 0.29105445 0.6148909
97 383 Germany: SPD DE 1953 text_only_direct_or_nearby Cultural 0.41201424225 0.239960775 0.59720855
98 383 Germany: SPD DE 1957 text_only_direct_or_nearby Cultural 0.361933815225 0.2020638 0.530546125
99 383 Germany: SPD DE 1961 text_only_direct_or_nearby Cultural 0.3403875836625 0.189634975 0.502483725
100 383 Germany: SPD DE 1965 text_only_direct_or_nearby Cultural 0.349500525125 0.197610625 0.519782425
101 383 Germany: SPD DE 1969 text_only_direct_or_nearby Cultural 0.367453271375 0.2312429 0.51644905
102 383 Germany: SPD DE 1972 both_direct_or_nearby Cultural 0.370199619125 0.2680677 0.473299475
103 383 Germany: SPD DE 1976 both_direct_or_nearby Cultural 0.28787436475 0.21332925 0.364130925
104 383 Germany: SPD DE 1980 both_direct_or_nearby Cultural 0.346500680875 0.245533525 0.448428875
105 383 Germany: SPD DE 1983 both_direct_or_nearby Cultural 0.3643932085 0.2585273 0.474535775
106 383 Germany: SPD DE 1987 both_direct_or_nearby Cultural 0.373242374375 0.266124675 0.479610175
107 383 Germany: SPD DE 1990 both_direct_or_nearby Cultural 0.354967389625 0.25200245 0.460687325
108 383 Germany: SPD DE 1994 both_direct_or_nearby Cultural 0.410000861625 0.3313879 0.488224525
109 383 Germany: SPD DE 1998 both_direct_or_nearby Cultural 0.458140040875 0.389836225 0.525399375
110 383 Germany: SPD DE 2002 both_direct_or_nearby Cultural 0.442940016 0.3876151 0.4964075
111 383 Germany: SPD DE 2005 both_direct_or_nearby Cultural 0.456470195875 0.39176415 0.5189629
112 383 Germany: SPD DE 2009 both_direct_or_nearby Cultural 0.389663805625 0.32557465 0.451253625
113 383 Germany: SPD DE 2013 both_direct_or_nearby Cultural 0.2109102855 0.15503785 0.268407625
114 383 Germany: SPD DE 2017 both_direct_or_nearby Cultural 0.40809383225 0.35346185 0.45866175
115 383 Germany: SPD DE 2021 both_direct_or_nearby Cultural 0.36824540875 0.27656695 0.461600275
116 383 Germany: SPD DE 2025 both_direct_or_nearby Cultural 0.3699090585 0.27155195 0.46999925
117 379 Denmark: Social Democrats DK 1945 text_only_direct_or_nearby Cultural 0.367253073125 0.222935775 0.513292475
118 379 Denmark: Social Democrats DK 1947 text_only_direct_or_nearby Cultural 0.38745918675 0.234832075 0.545806525
119 379 Denmark: Social Democrats DK 1950 text_only_direct_or_nearby Cultural 0.3968814625 0.2531828 0.544707225
120 379 Denmark: Social Democrats DK 1953 text_only_direct_or_nearby Cultural 0.42331649125 0.283525625 0.5604742
121 379 Denmark: Social Democrats DK 1957 text_only_direct_or_nearby Cultural 0.42351804925 0.268476675 0.589941825
122 379 Denmark: Social Democrats DK 1960 text_only_direct_or_nearby Cultural 0.384011194875 0.24432495 0.52920335
123 379 Denmark: Social Democrats DK 1964 text_only_direct_or_nearby Cultural 0.362245514375 0.224588175 0.502448675
124 379 Denmark: Social Democrats DK 1966 text_only_direct_or_nearby Cultural 0.3679157005 0.230511325 0.5031338
125 379 Denmark: Social Democrats DK 1968 text_only_direct_or_nearby Cultural 0.382175081625 0.25659165 0.507958075
126 379 Denmark: Social Democrats DK 1971 both_direct_or_nearby Cultural 0.470225764375 0.36279785 0.577130125
127 379 Denmark: Social Democrats DK 1973 both_direct_or_nearby Cultural 0.501466689875 0.3986878 0.60298055
128 379 Denmark: Social Democrats DK 1975 both_direct_or_nearby Cultural 0.51072920025 0.41364105 0.608167125
129 379 Denmark: Social Democrats DK 1977 both_direct_or_nearby Cultural 0.48888551625 0.3865325 0.588461525
130 379 Denmark: Social Democrats DK 1979 both_direct_or_nearby Cultural 0.469331547625 0.36461055 0.5720459
131 379 Denmark: Social Democrats DK 1981 both_direct_or_nearby Cultural 0.440505821125 0.334911475 0.544872125
132 379 Denmark: Social Democrats DK 1984 both_direct_or_nearby Cultural 0.427927746125 0.3199225 0.535149325
133 379 Denmark: Social Democrats DK 1987 both_direct_or_nearby Cultural 0.42510295375 0.332117075 0.518948075
134 379 Denmark: Social Democrats DK 1988 both_direct_or_nearby Cultural 0.4189467795 0.3290549 0.51075845
135 379 Denmark: Social Democrats DK 1990 both_direct_or_nearby Cultural 0.394384283125 0.289929725 0.498693825
136 379 Denmark: Social Democrats DK 1994 both_direct_or_nearby Cultural 0.365051991 0.259654975 0.469250275
137 379 Denmark: Social Democrats DK 1998 both_direct_or_nearby Cultural 0.4057949125 0.318039475 0.493913625
138 379 Denmark: Social Democrats DK 2001 both_direct_or_nearby Cultural 0.377267655375 0.2986888 0.456555875
139 379 Denmark: Social Democrats DK 2005 both_direct_or_nearby Cultural 0.3840174395 0.296652025 0.472492425
140 379 Denmark: Social Democrats DK 2007 both_direct_or_nearby Cultural 0.377073825875 0.288222975 0.46642125
141 379 Denmark: Social Democrats DK 2011 both_direct_or_nearby Cultural 0.44217709925 0.351102525 0.5332076
142 379 Denmark: Social Democrats DK 2015 both_direct_or_nearby Cultural 0.5274942445 0.44580275 0.608968575
143 379 Denmark: Social Democrats DK 2019 both_direct_or_nearby Cultural 0.440971318625 0.38293125 0.4998911
144 409 Sweden: Sweden Democrats SE 2010 both_direct_or_nearby Cultural 0.721373298875 0.653960175 0.78768305
145 409 Sweden: Sweden Democrats SE 2014 both_direct_or_nearby Cultural 0.750375772125 0.690054075 0.81067825
146 409 Sweden: Sweden Democrats SE 2018 both_direct_or_nearby Cultural 0.671777804625 0.5936524 0.747156725
147 409 Sweden: Sweden Democrats SE 2022 both_direct_or_nearby Cultural 0.753425659 0.653108875 0.84672335

Some files were not shown because too many files have changed in this diff Show More